{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":10221521,"sourceType":"datasetVersion","datasetId":6318833},{"sourceId":10343758,"sourceType":"datasetVersion","datasetId":6405356}],"dockerImageVersionId":30627,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This is a notebook in progress whose purpose is to change a semi generic notebook that Ive used for at least the last 6 comps and  to improve  or change the code  to make it more streamlined and general, so that I can quickly press \"run all\" having input only a few variables such as dataset and target feature. - I guess this is my attempt at an autoML, but with learning as part of the process.\nThere is very little eda currently other than shepperding all the features into categoricals, assigning numerals to nan, imputing values for the floats and pulling apart the date feature\nThere are still a lot of things to do, not to mention a lot of code annotation. Currently \"runall\" should take you to the section that begins XGBoost classifier. From there you are on your own journey of hyper param tuning, ensembling, stacking etc....","metadata":{}},{"cell_type":"markdown","source":"\n","metadata":{}},{"cell_type":"markdown","source":"# **Binary mental health  - sk models**","metadata":{}},{"cell_type":"code","source":"\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \nimport gc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:11.434004Z","iopub.execute_input":"2025-01-01T01:26:11.434991Z","iopub.status.idle":"2025-01-01T01:26:11.481257Z","shell.execute_reply.started":"2025-01-01T01:26:11.434951Z","shell.execute_reply":"2025-01-01T01:26:11.480259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import scipy\n\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:11.483298Z","iopub.execute_input":"2025-01-01T01:26:11.484165Z","iopub.status.idle":"2025-01-01T01:26:12.406293Z","shell.execute_reply.started":"2025-01-01T01:26:11.484121Z","shell.execute_reply":"2025-01-01T01:26:12.405274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n#!pip install --upgrade scikit-learn\n!pip install -q scikit-learn==1.5.2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:12.407374Z","iopub.execute_input":"2025-01-01T01:26:12.407785Z","iopub.status.idle":"2025-01-01T01:26:28.18667Z","shell.execute_reply.started":"2025-01-01T01:26:12.407756Z","shell.execute_reply":"2025-01-01T01:26:28.185328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n#creating deepcopy of model instances\nfrom copy import deepcopy\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.tree import export_graphviz\nimport graphviz\nimport pydot\nfrom IPython.display import Image\nfrom sklearn.model_selection import RepeatedStratifiedKFold\nfrom sklearn.metrics import accuracy_score,f1_score,roc_auc_score,confusion_matrix,roc_curve\nfrom skopt import BayesSearchCV\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.preprocessing import PowerTransformer\nimport time\nimport sklearn\nfrom scipy.stats import randint as sp_randint\nfrom scipy.stats import uniform as sp_uniform","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:28.190125Z","iopub.execute_input":"2025-01-01T01:26:28.190856Z","iopub.status.idle":"2025-01-01T01:26:28.558907Z","shell.execute_reply.started":"2025-01-01T01:26:28.190817Z","shell.execute_reply":"2025-01-01T01:26:28.558084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!pip install xgboost\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:28.560104Z","iopub.execute_input":"2025-01-01T01:26:28.560562Z","iopub.status.idle":"2025-01-01T01:26:28.564597Z","shell.execute_reply.started":"2025-01-01T01:26:28.560533Z","shell.execute_reply":"2025-01-01T01:26:28.563703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\nfrom xgboost import XGBRegressor\n#from xgboost import XGBRegressor\nfrom xgboost import XGBClassifier","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:28.565788Z","iopub.execute_input":"2025-01-01T01:26:28.566086Z","iopub.status.idle":"2025-01-01T01:26:28.748305Z","shell.execute_reply.started":"2025-01-01T01:26:28.566057Z","shell.execute_reply":"2025-01-01T01:26:28.747367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!pip install lightgbm\nimport lightgbm as lgb\nfrom lightgbm import LGBMClassifier\nfrom lightgbm import LGBMRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:28.749447Z","iopub.execute_input":"2025-01-01T01:26:28.749755Z","iopub.status.idle":"2025-01-01T01:26:29.807254Z","shell.execute_reply.started":"2025-01-01T01:26:28.749727Z","shell.execute_reply":"2025-01-01T01:26:29.806392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import catboost as cat\nfrom catboost import CatBoostClassifier\nfrom catboost import CatBoostRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:29.808477Z","iopub.execute_input":"2025-01-01T01:26:29.808783Z","iopub.status.idle":"2025-01-01T01:26:30.001723Z","shell.execute_reply.started":"2025-01-01T01:26:29.808755Z","shell.execute_reply":"2025-01-01T01:26:30.000718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##for the sklearn models\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.ensemble import HistGradientBoostingClassifier\n#from xgboost import XGBClassifier\n#from xgboost import XGBRegressor\n# This function displays the splits of the tree\nfrom sklearn.tree import plot_tree\n\nfrom sklearn.metrics import ConfusionMatrixDisplay, confusion_matrix\nfrom sklearn.metrics import recall_score, precision_score, f1_score, accuracy_score\n# Import GridSearchCV\nfrom sklearn.model_selection import GridSearchCV, RandomizedSearchCV\nfrom sklearn.base import clone","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:30.002863Z","iopub.execute_input":"2025-01-01T01:26:30.003183Z","iopub.status.idle":"2025-01-01T01:26:30.008763Z","shell.execute_reply.started":"2025-01-01T01:26:30.003156Z","shell.execute_reply":"2025-01-01T01:26:30.007837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!pip install shap\n#import shap","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:30.01344Z","iopub.execute_input":"2025-01-01T01:26:30.014145Z","iopub.status.idle":"2025-01-01T01:26:30.025936Z","shell.execute_reply.started":"2025-01-01T01:26:30.014109Z","shell.execute_reply":"2025-01-01T01:26:30.024956Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load a dataset into a Pandas Dataframe\ndataset_df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))\n#train_df = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntest_df = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\nprint(\"Full test dataset shape is {}\".format(test_df.shape))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:30.027324Z","iopub.execute_input":"2025-01-01T01:26:30.027746Z","iopub.status.idle":"2025-01-01T01:26:40.095206Z","shell.execute_reply.started":"2025-01-01T01:26:30.027709Z","shell.execute_reply":"2025-01-01T01:26:40.094176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TARGET = 'Premium Amount'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:40.096532Z","iopub.execute_input":"2025-01-01T01:26:40.096931Z","iopub.status.idle":"2025-01-01T01:26:40.102074Z","shell.execute_reply.started":"2025-01-01T01:26:40.096893Z","shell.execute_reply":"2025-01-01T01:26:40.100941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ID = 'id'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:40.103449Z","iopub.execute_input":"2025-01-01T01:26:40.103751Z","iopub.status.idle":"2025-01-01T01:26:40.113332Z","shell.execute_reply.started":"2025-01-01T01:26:40.103723Z","shell.execute_reply":"2025-01-01T01:26:40.112343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:40.114696Z","iopub.execute_input":"2025-01-01T01:26:40.115327Z","iopub.status.idle":"2025-01-01T01:26:40.158749Z","shell.execute_reply.started":"2025-01-01T01:26:40.115297Z","shell.execute_reply":"2025-01-01T01:26:40.157851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_df.describe(include='all')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:40.160126Z","iopub.execute_input":"2025-01-01T01:26:40.160383Z","iopub.status.idle":"2025-01-01T01:26:42.862267Z","shell.execute_reply.started":"2025-01-01T01:26:40.16036Z","shell.execute_reply":"2025-01-01T01:26:42.861235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_df.describe().style.background_gradient()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:42.863492Z","iopub.execute_input":"2025-01-01T01:26:42.863775Z","iopub.status.idle":"2025-01-01T01:26:43.594247Z","shell.execute_reply.started":"2025-01-01T01:26:42.863749Z","shell.execute_reply":"2025-01-01T01:26:43.593236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n = dataset_df.nunique(axis=0) \nprint(\"No.of.unique values :\",  n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:43.595589Z","iopub.execute_input":"2025-01-01T01:26:43.596581Z","iopub.status.idle":"2025-01-01T01:26:44.784126Z","shell.execute_reply.started":"2025-01-01T01:26:43.596548Z","shell.execute_reply":"2025-01-01T01:26:44.783084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"t = test_df.nunique(axis=0) \nprint(\"No.of.unique values :\",  t)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:44.78538Z","iopub.execute_input":"2025-01-01T01:26:44.785715Z","iopub.status.idle":"2025-01-01T01:26:45.545325Z","shell.execute_reply.started":"2025-01-01T01:26:44.785686Z","shell.execute_reply":"2025-01-01T01:26:45.544288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#make this into a dataset for later access?\n'''\ndef print_unique_values(test_dataset, ):\n    try:\n        \n        for column in test_dataset.columns:\n            unique_values = test_dataset[column].unique()[:7]  # Taking at least 7 unique values\n            unique_values_str = ', '.join(map(str, unique_values))\n            data_type = test_dataset[column].dtype\n            \n\n    \n    except Exception as e:\n        print_error(str(e))\n  '''      ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:45.546836Z","iopub.execute_input":"2025-01-01T01:26:45.547553Z","iopub.status.idle":"2025-01-01T01:26:45.554147Z","shell.execute_reply.started":"2025-01-01T01:26:45.547512Z","shell.execute_reply":"2025-01-01T01:26:45.553109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:45.555599Z","iopub.execute_input":"2025-01-01T01:26:45.556006Z","iopub.status.idle":"2025-01-01T01:26:46.173871Z","shell.execute_reply.started":"2025-01-01T01:26:45.55597Z","shell.execute_reply":"2025-01-01T01:26:46.172876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:46.174949Z","iopub.execute_input":"2025-01-01T01:26:46.175266Z","iopub.status.idle":"2025-01-01T01:26:46.786294Z","shell.execute_reply.started":"2025-01-01T01:26:46.175238Z","shell.execute_reply":"2025-01-01T01:26:46.785276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:46.787788Z","iopub.execute_input":"2025-01-01T01:26:46.788214Z","iopub.status.idle":"2025-01-01T01:26:47.222972Z","shell.execute_reply.started":"2025-01-01T01:26:46.788178Z","shell.execute_reply":"2025-01-01T01:26:47.221908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"duplicates = dataset_df[dataset_df.duplicated()]\nduplicates","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:47.224251Z","iopub.execute_input":"2025-01-01T01:26:47.224549Z","iopub.status.idle":"2025-01-01T01:26:49.143695Z","shell.execute_reply.started":"2025-01-01T01:26:47.224524Z","shell.execute_reply":"2025-01-01T01:26:49.142764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.kdeplot(data=dataset_df, x='Premium Amount', color='orange', fill=True);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:49.144969Z","iopub.execute_input":"2025-01-01T01:26:49.145399Z","iopub.status.idle":"2025-01-01T01:26:54.327802Z","shell.execute_reply.started":"2025-01-01T01:26:49.145362Z","shell.execute_reply":"2025-01-01T01:26:54.326797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dropped_df = dataset_df.copy()\ndropped_df.drop([ID], axis=1, inplace=True)\n\ndropped_test_df = test_df.copy()\ndropped_test_df.drop([ID], axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:54.329222Z","iopub.execute_input":"2025-01-01T01:26:54.330276Z","iopub.status.idle":"2025-01-01T01:26:54.908962Z","shell.execute_reply.started":"2025-01-01T01:26:54.330234Z","shell.execute_reply":"2025-01-01T01:26:54.908128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_cols = dropped_test_df.columns\nall_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:54.910104Z","iopub.execute_input":"2025-01-01T01:26:54.9104Z","iopub.status.idle":"2025-01-01T01:26:54.916606Z","shell.execute_reply.started":"2025-01-01T01:26:54.910373Z","shell.execute_reply":"2025-01-01T01:26:54.915782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#pd.set_option('display.max_rows', None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:54.917681Z","iopub.execute_input":"2025-01-01T01:26:54.917975Z","iopub.status.idle":"2025-01-01T01:26:54.925464Z","shell.execute_reply.started":"2025-01-01T01:26:54.91795Z","shell.execute_reply":"2025-01-01T01:26:54.924692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#delete rows with nan in trial for vehicleAge and Insurance duration\ndef remove_rows_with_nan(df, columns_to_check):\n    \"\"\"\n    Removes rows from a Pandas DataFrame where any of the specified columns contain NaN values.\n\n    Args:\n        df: The Pandas DataFrame.\n        columns_to_check: A list of column names to check for NaN values.\n\n    Returns:\n        A new Pandas DataFrame with rows containing NaN in the specified columns removed.  Returns None if input is invalid.\n    \"\"\"\n    if not isinstance(df, pd.DataFrame):\n        print(\"Error: Input must be a Pandas DataFrame.\")\n        return None\n    if not isinstance(columns_to_check, list) or not all(isinstance(col, str) for col in columns_to_check):\n        print(\"Error: columns_to_check must be a list of strings.\")\n        return None\n    if not all(col in df.columns for col in columns_to_check):\n        print(\"Error: Not all columns in columns_to_check exist in the DataFrame.\")\n        return None\n\n\n    initial_rows = len(df)\n    print(f\"Initial number of rows: {initial_rows}\")\n\n    #Efficiently check for NaNs in specified columns and drop rows\n    df_cleaned = df.dropna(subset=columns_to_check)\n\n    final_rows = len(df_cleaned)\n    print(f\"Number of rows after removing rows with NaN: {final_rows}\")\n\n    return df_cleaned\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:54.932552Z","iopub.execute_input":"2025-01-01T01:26:54.932921Z","iopub.status.idle":"2025-01-01T01:26:54.940421Z","shell.execute_reply.started":"2025-01-01T01:26:54.932883Z","shell.execute_reply":"2025-01-01T01:26:54.939407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = dropped_df.copy()\n#columns_to_remove_nan = ['Vehicle Age', 'Insurance Duration'] #Specify which columns to check for NaNs\n#X = remove_rows_with_nan(X,columns_to_remove_nan)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:54.94146Z","iopub.execute_input":"2025-01-01T01:26:54.941727Z","iopub.status.idle":"2025-01-01T01:26:55.113158Z","shell.execute_reply.started":"2025-01-01T01:26:54.941703Z","shell.execute_reply":"2025-01-01T01:26:55.112076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nY = X[TARGET]\n#Y = LabelEncoder().fit_transform(y)\nY","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:55.114498Z","iopub.execute_input":"2025-01-01T01:26:55.115244Z","iopub.status.idle":"2025-01-01T01:26:55.123603Z","shell.execute_reply.started":"2025-01-01T01:26:55.115209Z","shell.execute_reply":"2025-01-01T01:26:55.122514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#stop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:55.125222Z","iopub.execute_input":"2025-01-01T01:26:55.125524Z","iopub.status.idle":"2025-01-01T01:26:55.13234Z","shell.execute_reply.started":"2025-01-01T01:26:55.125497Z","shell.execute_reply":"2025-01-01T01:26:55.131518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\nX.drop([TARGET], axis=1, inplace=True)\nX.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:55.133432Z","iopub.execute_input":"2025-01-01T01:26:55.133718Z","iopub.status.idle":"2025-01-01T01:26:55.951261Z","shell.execute_reply.started":"2025-01-01T01:26:55.133693Z","shell.execute_reply":"2025-01-01T01:26:55.950204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nnum_cols = X.select_dtypes(include=np.number).columns.tolist()\nnum_cols\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:55.952501Z","iopub.execute_input":"2025-01-01T01:26:55.952799Z","iopub.status.idle":"2025-01-01T01:26:55.991659Z","shell.execute_reply.started":"2025-01-01T01:26:55.952773Z","shell.execute_reply":"2025-01-01T01:26:55.990845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"obj_cols = X.select_dtypes(include='object').columns.tolist()\nobj_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:55.992933Z","iopub.execute_input":"2025-01-01T01:26:55.993346Z","iopub.status.idle":"2025-01-01T01:26:56.154208Z","shell.execute_reply.started":"2025-01-01T01:26:55.993309Z","shell.execute_reply":"2025-01-01T01:26:56.153112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\ndef print_feature_by_target(the_cols, the_outcome, the_dataset):\n    for col in the_cols:\n       \n        df0 = the_dataset.groupby([col, the_outcome])[the_outcome].count().unstack()\n        print(df0)       \n        print('######################################\\n')\n        '''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:56.155463Z","iopub.execute_input":"2025-01-01T01:26:56.15588Z","iopub.status.idle":"2025-01-01T01:26:56.166111Z","shell.execute_reply.started":"2025-01-01T01:26:56.155842Z","shell.execute_reply":"2025-01-01T01:26:56.165072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#print_feature_by_target(num_cols, outcome_col, dropped_df)\n#print_feature_by_target(num_cols, TARGET, dropped_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:56.167321Z","iopub.execute_input":"2025-01-01T01:26:56.167658Z","iopub.status.idle":"2025-01-01T01:26:56.176133Z","shell.execute_reply.started":"2025-01-01T01:26:56.167623Z","shell.execute_reply":"2025-01-01T01:26:56.175289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#stop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:56.177261Z","iopub.execute_input":"2025-01-01T01:26:56.177633Z","iopub.status.idle":"2025-01-01T01:26:56.185443Z","shell.execute_reply.started":"2025-01-01T01:26:56.177606Z","shell.execute_reply":"2025-01-01T01:26:56.184555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def print_train_count(the_cols,  the_dataset):\n    for col in the_cols:\n        #print(col)\n        #df0 = the_dataset.groupby([col, the_outcome])[the_outcome].count().unstack()\n        df0 = the_dataset[col].value_counts(normalize=True)\n        print(df0)\n        print('######################################\\n')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:56.186713Z","iopub.execute_input":"2025-01-01T01:26:56.187535Z","iopub.status.idle":"2025-01-01T01:26:56.195855Z","shell.execute_reply.started":"2025-01-01T01:26:56.187496Z","shell.execute_reply":"2025-01-01T01:26:56.194915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#pd.set_option('display.max_rows', None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:56.197235Z","iopub.execute_input":"2025-01-01T01:26:56.197895Z","iopub.status.idle":"2025-01-01T01:26:56.207777Z","shell.execute_reply.started":"2025-01-01T01:26:56.197856Z","shell.execute_reply":"2025-01-01T01:26:56.206813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print_train_count(num_cols,  dropped_test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:56.208991Z","iopub.execute_input":"2025-01-01T01:26:56.209361Z","iopub.status.idle":"2025-01-01T01:26:56.402059Z","shell.execute_reply.started":"2025-01-01T01:26:56.209331Z","shell.execute_reply":"2025-01-01T01:26:56.401066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\ndef print_test_count(the_cols,  the_dataset):\n    for col in the_cols:\n\n        #df0 = dataset_df[col].value_counts(normalize=True)\n        df0 = the_dataset[col].value_counts()\n        print(df0)\n        print('######################################\\n')\n '''       ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:56.403222Z","iopub.execute_input":"2025-01-01T01:26:56.403526Z","iopub.status.idle":"2025-01-01T01:26:56.409861Z","shell.execute_reply.started":"2025-01-01T01:26:56.4035Z","shell.execute_reply":"2025-01-01T01:26:56.40884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print_train_count(obj_cols,  dropped_test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:56.41114Z","iopub.execute_input":"2025-01-01T01:26:56.411442Z","iopub.status.idle":"2025-01-01T01:26:57.373398Z","shell.execute_reply.started":"2025-01-01T01:26:56.411417Z","shell.execute_reply":"2025-01-01T01:26:57.372416Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def print_unique_vals_by_col(the_cols, the_dataset):\n    for col in the_cols:\n        df0 =the_dataset[col].unique()\n        print(col)\n        print('\\n')\n        print(df0)\n        print('######################################\\n')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:57.374649Z","iopub.execute_input":"2025-01-01T01:26:57.37495Z","iopub.status.idle":"2025-01-01T01:26:57.380255Z","shell.execute_reply.started":"2025-01-01T01:26:57.374923Z","shell.execute_reply":"2025-01-01T01:26:57.379308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#or from genai\ndef list_unique_entries(df):\n    for col in df.columns:\n        unique_entries = df[col].unique()\n        print(f\"Unique entries for {col}: {unique_entries}\")\n        print('--------------------------------------------\\n')\n        print('--------------------------------------------\\n')\n    return unique_entries\n#@list_unique_entries(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:57.38167Z","iopub.execute_input":"2025-01-01T01:26:57.38207Z","iopub.status.idle":"2025-01-01T01:26:57.392937Z","shell.execute_reply.started":"2025-01-01T01:26:57.382014Z","shell.execute_reply":"2025-01-01T01:26:57.392075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"list_unique_entries(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:57.394377Z","iopub.execute_input":"2025-01-01T01:26:57.395192Z","iopub.status.idle":"2025-01-01T01:26:58.521623Z","shell.execute_reply.started":"2025-01-01T01:26:57.395152Z","shell.execute_reply":"2025-01-01T01:26:58.520615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#or list by frequency:\ndef list_unique_entries_by_freq(df):\n    for col in df.columns:\n        unique_counts = df[col].value_counts()\n        print(f\"Unique entries for {col}, sorted by frequency (descending):\\n{unique_counts}\")\n        print('--------------------------------------------\\n')\n        print('--------------------------------------------\\n')\n    return unique_counts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:58.523074Z","iopub.execute_input":"2025-01-01T01:26:58.523511Z","iopub.status.idle":"2025-01-01T01:26:58.529053Z","shell.execute_reply.started":"2025-01-01T01:26:58.523473Z","shell.execute_reply":"2025-01-01T01:26:58.528111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#object to numeral\n#start convert date time col to unix timestamp?\nX['sec_since_70'] = pd.to_datetime(X['Policy Start Date'])#.astype(int)\nX['sec_since_70'] = (X['sec_since_70'] - pd.Timestamp(\"1970-01-01\")) // pd.Timedelta('1s') #without this im doing ns\ndropped_test_df['sec_since_70'] = pd.to_datetime(dropped_test_df['Policy Start Date'])#.astype(int)\ndropped_test_df['sec_since_70'] = (dropped_test_df['sec_since_70'] - pd.Timestamp(\"1970-01-01\")) // pd.Timedelta('1s') ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:58.530349Z","iopub.execute_input":"2025-01-01T01:26:58.530668Z","iopub.status.idle":"2025-01-01T01:26:59.273798Z","shell.execute_reply.started":"2025-01-01T01:26:58.530643Z","shell.execute_reply":"2025-01-01T01:26:59.272931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#add a nanosec col\n#X['nano_sec_since_70'] = pd.to_datetime(X['Policy Start Date']).astype(int)\n\n#dropped_test_df['nano_sec_since_70'] = pd.to_datetime(dropped_test_df['Policy Start Date']).astype(int)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:59.274959Z","iopub.execute_input":"2025-01-01T01:26:59.275296Z","iopub.status.idle":"2025-01-01T01:26:59.279437Z","shell.execute_reply.started":"2025-01-01T01:26:59.275268Z","shell.execute_reply":"2025-01-01T01:26:59.278536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#separate to year, mon, day, hour, min, sec, nanosecs cols\ndef split_datetime_column(df, datetime_column):\n    \"\"\"\n    Splits a datetime column in a Pandas DataFrame into separate columns for year, month, day, hour, minute, second, and nanosecond.\n\n    Args:\n        df: The Pandas DataFrame.\n        datetime_column: The name of the column containing datetime objects.\n\n    Returns:\n        A new Pandas DataFrame with the original column replaced by the split components.\n    \"\"\"\n    df_copy = df.copy() # Create a copy to avoid modifying the original df\n\n    if datetime_column not in df_copy.columns:\n      print(f\"Column '{datetime_column}' not found in DataFrame.\")\n      return df_copy # return the same dataframe\n\n    # Convert to datetime objects if it's not already\n    df_copy[datetime_column] = pd.to_datetime(df_copy[datetime_column])\n\n    # Extract the components\n    df_copy['_year'] = df_copy[datetime_column].dt.year\n    df_copy['_month'] = df_copy[datetime_column].dt.month\n    df_copy['_day'] = df_copy[datetime_column].dt.day\n    df_copy['_d_of_w'] = df_copy[datetime_column].dt.day_of_week\n    #df_copy['_w_of_y'] = df_copy[datetime_column].dt.isocalendar().week\n    #df_copy['_hour'] = df_copy[datetime_column].dt.hour\n    #df_copy['_minute'] = df_copy[datetime_column].dt.minute\n    #df_copy['_second'] = df_copy[datetime_column].dt.second\n    #df_copy['_nanosecond'] = df_copy[datetime_column].dt.nanosecond\n\n    # Optionally drop the original datetime column, if needed.\n    # df_copy = df_copy.drop(columns=[datetime_column]) # added comment in case the user wants to keep the old column\n\n    return df_copy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:59.280684Z","iopub.execute_input":"2025-01-01T01:26:59.281092Z","iopub.status.idle":"2025-01-01T01:26:59.290499Z","shell.execute_reply.started":"2025-01-01T01:26:59.281045Z","shell.execute_reply":"2025-01-01T01:26:59.289549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = split_datetime_column(X, 'Policy Start Date')\ndropped_test_df = split_datetime_column(dropped_test_df, 'Policy Start Date')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:26:59.291764Z","iopub.execute_input":"2025-01-01T01:26:59.292111Z","iopub.status.idle":"2025-01-01T01:27:01.277515Z","shell.execute_reply.started":"2025-01-01T01:26:59.29208Z","shell.execute_reply":"2025-01-01T01:27:01.27636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dropped_test_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:01.278951Z","iopub.execute_input":"2025-01-01T01:27:01.279418Z","iopub.status.idle":"2025-01-01T01:27:01.668444Z","shell.execute_reply.started":"2025-01-01T01:27:01.279371Z","shell.execute_reply":"2025-01-01T01:27:01.667484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.head(30)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:01.670808Z","iopub.execute_input":"2025-01-01T01:27:01.671319Z","iopub.status.idle":"2025-01-01T01:27:01.707849Z","shell.execute_reply.started":"2025-01-01T01:27:01.671287Z","shell.execute_reply":"2025-01-01T01:27:01.706815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:01.709164Z","iopub.execute_input":"2025-01-01T01:27:01.7095Z","iopub.status.idle":"2025-01-01T01:27:02.580897Z","shell.execute_reply.started":"2025-01-01T01:27:01.709471Z","shell.execute_reply":"2025-01-01T01:27:02.57981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dropped_test_df.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:02.582044Z","iopub.execute_input":"2025-01-01T01:27:02.582349Z","iopub.status.idle":"2025-01-01T01:27:03.161948Z","shell.execute_reply.started":"2025-01-01T01:27:02.582322Z","shell.execute_reply":"2025-01-01T01:27:03.160962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#stop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:03.163164Z","iopub.execute_input":"2025-01-01T01:27:03.163483Z","iopub.status.idle":"2025-01-01T01:27:03.1676Z","shell.execute_reply.started":"2025-01-01T01:27:03.163455Z","shell.execute_reply":"2025-01-01T01:27:03.166576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\ndef count_rows_with_number(df, the_value, the_column_name):\n    \"\"\"\n    Counts the number of rows in a DataFrame where the column equals a specified value.\n\n    Args:\n      df: The Pandas DataFrame.\n      the_value: The integer value of the column to filter on.\n      the_column_name: The name of the column containing  values.\n\n    Returns:\n      The number of rows where the  column equals the specified value.\n    \"\"\"\n\n    if the_column_name not in df.columns:\n      print(f\"Column '{the_column_name}' not found in DataFrame.\")\n      return 0 # Return 0 if hour_column_name does not exist in the DF\n\n    try:\n      count = (df[the_column_name] == the_value).sum()\n      return count\n    except TypeError:\n      print(f\"Ensure the column '{hour_column_name}' contains integers.\")\n      return 0\n\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:03.168756Z","iopub.execute_input":"2025-01-01T01:27:03.169059Z","iopub.status.idle":"2025-01-01T01:27:03.181937Z","shell.execute_reply.started":"2025-01-01T01:27:03.169005Z","shell.execute_reply":"2025-01-01T01:27:03.180945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nwhat_to_count = 15\ncol_to_count = '_hour'\ncount = count_rows_with_number(X, what_to_count, col_to_count)\nprint(count)\n#\t1199993.0/800000  #remember I dropp th 7 X that a nans\n#all of dropped df hour, minutre, second, nanosecond are : 15:21:39:0\n#X samne all 15:21:39:0\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:03.183233Z","iopub.execute_input":"2025-01-01T01:27:03.183577Z","iopub.status.idle":"2025-01-01T01:27:03.197352Z","shell.execute_reply.started":"2025-01-01T01:27:03.18355Z","shell.execute_reply":"2025-01-01T01:27:03.196299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\ndef delete_columns(df, columns_to_delete):\n    \"\"\"\n    Deletes specified columns from a Pandas DataFrame.\n\n    Args:\n        df: The Pandas DataFrame.\n        columns_to_delete: A list of strings, representing the column names to delete.\n\n    Returns:\n        A tuple containing:\n          - A boolean: True for success, False for failure.\n          - A string: A success or failure message.\n          - A dataframe: if the removal is successful, the transformed dataframe is returned. If failure it returns the original.\n    \"\"\"\n\n    df_copy = df.copy()  # Create a copy to avoid modifying the original df\n    try:\n      columns_removed = []\n      for col in columns_to_delete:\n          if col in df_copy.columns:\n              df_copy = df_copy.drop(columns=[col])\n              columns_removed.append(col) #track successfully removed columns\n          else:\n              print(f\"Column '{col}' not found in DataFrame.\")\n\n      if len(columns_removed) > 0 :\n            return True, f\"Successfully removed columns: {', '.join(columns_removed)}\", df_copy\n      else:\n            return False, \"No columns were successfully removed from the list provided.\", df_copy\n\n    except Exception as e:\n        return False, f\"An error occurred during column deletion: {e}\", df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:03.19867Z","iopub.execute_input":"2025-01-01T01:27:03.199084Z","iopub.status.idle":"2025-01-01T01:27:03.209739Z","shell.execute_reply.started":"2025-01-01T01:27:03.199043Z","shell.execute_reply":"2025-01-01T01:27:03.208798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns_to_remove = ['Policy Start Date', ]\nsuccess, message, X = delete_columns(X, columns_to_remove)\n    \nprint(f\"Success: {success}, Message: {message}\")\n\nsuccess, message, dropped_test_df = delete_columns(dropped_test_df, columns_to_remove)\n    \nprint(f\"Success: {success}, Message: {message}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:03.211666Z","iopub.execute_input":"2025-01-01T01:27:03.212094Z","iopub.status.idle":"2025-01-01T01:27:04.417218Z","shell.execute_reply.started":"2025-01-01T01:27:03.212054Z","shell.execute_reply":"2025-01-01T01:27:04.4162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nfloat_cols =['Age', 'Number of Dependents', 'Vehicle Age', 'Previous Claims', 'Insurance Duration' ]\n#if I want to do this conversion I have to fill na , pick 50 since its \nX[float_cols] = X[float_cols].fillna(100).astype(int)\ndropped_test_df[float_cols] = dropped_test_df[float_cols].fillna(100).astype(int)\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:04.418391Z","iopub.execute_input":"2025-01-01T01:27:04.418705Z","iopub.status.idle":"2025-01-01T01:27:04.424881Z","shell.execute_reply.started":"2025-01-01T01:27:04.418676Z","shell.execute_reply":"2025-01-01T01:27:04.423848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nan_counts_X = X.isna().sum() \nnan_counts_test = dropped_test_df.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:04.42595Z","iopub.execute_input":"2025-01-01T01:27:04.42626Z","iopub.status.idle":"2025-01-01T01:27:05.327943Z","shell.execute_reply.started":"2025-01-01T01:27:04.426234Z","shell.execute_reply":"2025-01-01T01:27:05.326979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nan_counts_X","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:05.329104Z","iopub.execute_input":"2025-01-01T01:27:05.32941Z","iopub.status.idle":"2025-01-01T01:27:05.336402Z","shell.execute_reply.started":"2025-01-01T01:27:05.329381Z","shell.execute_reply":"2025-01-01T01:27:05.33532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nan_counts_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:05.337702Z","iopub.execute_input":"2025-01-01T01:27:05.338006Z","iopub.status.idle":"2025-01-01T01:27:05.34887Z","shell.execute_reply.started":"2025-01-01T01:27:05.337972Z","shell.execute_reply":"2025-01-01T01:27:05.348008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"high_nan_X = nan_counts_X[nan_counts_X > 50].index.tolist()\n#high_nan_test = nan_counts_test[nan_counts_test > 100].index.tolist()\nhigh_nan_test = nan_counts_test[nan_counts_test > 50].index.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:05.350058Z","iopub.execute_input":"2025-01-01T01:27:05.350358Z","iopub.status.idle":"2025-01-01T01:27:05.360689Z","shell.execute_reply.started":"2025-01-01T01:27:05.350331Z","shell.execute_reply":"2025-01-01T01:27:05.359725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nLrts try to do \nX_train['Annual Income'] = np.log1p(X_train['Annual Income'])\nX_test['Annual Income'] = np.log1p(X_test['Annual Income'])\nfor the 4 floats inciome, cred score, health score and time since 1970\n\nThen fill missing siumply:\ndef impute_missing_numerical_data(df):\n    num = df.select_dtypes(exclude='object').columns\n    for n in num:\n        df[n] = df[n].fillna(-1)\n\n    return df\n\ndef impute_missing_categorical_data(df):\n    cat = df.select_dtypes(include='object').columns\n    for c in cat:\n        df[c] = df[c].fillna('Unknown')\n\n    return df\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:05.361818Z","iopub.execute_input":"2025-01-01T01:27:05.362181Z","iopub.status.idle":"2025-01-01T01:27:05.377333Z","shell.execute_reply.started":"2025-01-01T01:27:05.362152Z","shell.execute_reply":"2025-01-01T01:27:05.376346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Deal with nan?\n#1. Delete rows in trial that have minimal nans: Vehicle age, insurance duration\n#2. in test give -1, -2 values to the missing in vehicle age, insurance duration\n#3. for high values lets:mmedian for floats, number for others but Age?  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:05.378908Z","iopub.execute_input":"2025-01-01T01:27:05.379336Z","iopub.status.idle":"2025-01-01T01:27:05.384301Z","shell.execute_reply.started":"2025-01-01T01:27:05.379296Z","shell.execute_reply":"2025-01-01T01:27:05.383384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n#delete rows with nan in trial for vehicleAge and Insurance duration\ndef remove_rows_with_nan(df, columns_to_check):\n    \"\"\"\n    Removes rows from a Pandas DataFrame where any of the specified columns contain NaN values.\n\n    Args:\n        df: The Pandas DataFrame.\n        columns_to_check: A list of column names to check for NaN values.\n\n    Returns:\n        A new Pandas DataFrame with rows containing NaN in the specified columns removed.  Returns None if input is invalid.\n    \"\"\"\n    if not isinstance(df, pd.DataFrame):\n        print(\"Error: Input must be a Pandas DataFrame.\")\n        return None\n    if not isinstance(columns_to_check, list) or not all(isinstance(col, str) for col in columns_to_check):\n        print(\"Error: columns_to_check must be a list of strings.\")\n        return None\n    if not all(col in df.columns for col in columns_to_check):\n        print(\"Error: Not all columns in columns_to_check exist in the DataFrame.\")\n        return None\n\n\n    initial_rows = len(df)\n    print(f\"Initial number of rows: {initial_rows}\")\n\n    #Efficiently check for NaNs in specified columns and drop rows\n    df_cleaned = df.dropna(subset=columns_to_check)\n\n    final_rows = len(df_cleaned)\n    print(f\"Number of rows after removing rows with NaN: {final_rows}\")\n\n    return df_cleaned\n\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:05.385454Z","iopub.execute_input":"2025-01-01T01:27:05.38575Z","iopub.status.idle":"2025-01-01T01:27:05.396844Z","shell.execute_reply.started":"2025-01-01T01:27:05.385725Z","shell.execute_reply":"2025-01-01T01:27:05.395839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#print(type(X))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:05.39806Z","iopub.execute_input":"2025-01-01T01:27:05.398361Z","iopub.status.idle":"2025-01-01T01:27:05.408643Z","shell.execute_reply.started":"2025-01-01T01:27:05.398335Z","shell.execute_reply":"2025-01-01T01:27:05.407775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#doing this before the X, y split above\n#columns_to_remove_nan = ['Vehicle Age', 'Insurance Duration'] #Specify which columns to check for NaNs\n#X = remove_rows_with_nan(X,columns_to_remove_nan)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:05.409884Z","iopub.execute_input":"2025-01-01T01:27:05.410555Z","iopub.status.idle":"2025-01-01T01:27:05.421261Z","shell.execute_reply.started":"2025-01-01T01:27:05.410523Z","shell.execute_reply":"2025-01-01T01:27:05.420251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# fill the above but in test with -1 and -2\n#or lets just use -1 as missing for all values\n#trying same fill val as others -100\nX['Vehicle Age'] = X['Vehicle Age'].fillna(-1)\n#X['Insurance Duration'] = X['Insurance Duration'].fillna(-2)\nX['Insurance Duration'] = X['Insurance Duration'].fillna(-1)\n\ndropped_test_df['Vehicle Age'] = dropped_test_df['Vehicle Age'].fillna(-1)\n#dropped_test_df['Insurance Duration'] = dropped_test_df['Insurance Duration'].fillna(-2)\ndropped_test_df['Insurance Duration'] = dropped_test_df['Insurance Duration'].fillna(-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:05.422407Z","iopub.execute_input":"2025-01-01T01:27:05.422704Z","iopub.status.idle":"2025-01-01T01:27:05.457663Z","shell.execute_reply.started":"2025-01-01T01:27:05.422677Z","shell.execute_reply":"2025-01-01T01:27:05.456623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#sns.kdeplot(data=dataset_df, x='Annual Income', color='orange', fill=True);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:05.45885Z","iopub.execute_input":"2025-01-01T01:27:05.459186Z","iopub.status.idle":"2025-01-01T01:27:05.463416Z","shell.execute_reply.started":"2025-01-01T01:27:05.459159Z","shell.execute_reply":"2025-01-01T01:27:05.462339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#sns.kdeplot(data=X, x='Annual Income', color='orange', fill=True);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:05.464726Z","iopub.execute_input":"2025-01-01T01:27:05.464986Z","iopub.status.idle":"2025-01-01T01:27:05.472116Z","shell.execute_reply.started":"2025-01-01T01:27:05.464963Z","shell.execute_reply":"2025-01-01T01:27:05.471216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#sns.kdeplot(data=dataset_df, x='Health Score', color='orange', fill=True);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:05.473439Z","iopub.execute_input":"2025-01-01T01:27:05.473715Z","iopub.status.idle":"2025-01-01T01:27:05.482112Z","shell.execute_reply.started":"2025-01-01T01:27:05.473691Z","shell.execute_reply":"2025-01-01T01:27:05.481212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#sns.kdeplot(data=X, x='Health Score', color='orange', fill=True);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:05.483243Z","iopub.execute_input":"2025-01-01T01:27:05.483549Z","iopub.status.idle":"2025-01-01T01:27:05.491249Z","shell.execute_reply.started":"2025-01-01T01:27:05.483523Z","shell.execute_reply":"2025-01-01T01:27:05.490297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#sns.kdeplot(data=dataset_df, x='Credit Score', color='orange', fill=True);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:05.492576Z","iopub.execute_input":"2025-01-01T01:27:05.492995Z","iopub.status.idle":"2025-01-01T01:27:05.499937Z","shell.execute_reply.started":"2025-01-01T01:27:05.492967Z","shell.execute_reply":"2025-01-01T01:27:05.499063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#sns.kdeplot(data=X, x='Credit Score', color='orange', fill=True);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:05.501094Z","iopub.execute_input":"2025-01-01T01:27:05.501367Z","iopub.status.idle":"2025-01-01T01:27:05.511407Z","shell.execute_reply.started":"2025-01-01T01:27:05.501343Z","shell.execute_reply":"2025-01-01T01:27:05.510456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#sns.kdeplot(data=X, x='sec_since_70', color='orange', fill=True);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:05.512502Z","iopub.execute_input":"2025-01-01T01:27:05.512767Z","iopub.status.idle":"2025-01-01T01:27:05.522591Z","shell.execute_reply.started":"2025-01-01T01:27:05.512743Z","shell.execute_reply":"2025-01-01T01:27:05.521729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#why not try log transform on these features since our target is bering log transformed?!\ndef log1p_transform_features(df, features):\n  \"\"\"\n  Applies np.log1p transformation to specified features in a Pandas DataFrame.\n\n  Args:\n    df: The Pandas DataFrame.\n    features: A list of strings, representing the column names to transform.\n\n  Returns:\n    A new Pandas DataFrame with the specified features transformed using np.log1p.\n  \"\"\"\n  df_copy = df.copy() # Create a copy to avoid modifying the original df\n  for feature in features:\n    if feature in df_copy.columns: # check to see if the feature is there\n      df_copy[feature] = np.log1p(df_copy[feature])\n    else:\n      print(f\"Feature '{feature}' not found in DataFrame columns.\")\n  return df_copy\n\nn_cols=['Annual Income', 'Health Score', 'Credit Score', 'sec_since_70' ]\n#n_cols=['Annual Income',  'sec_since_70' ]\nX = log1p_transform_features(X, n_cols)\ndropped_test_df = log1p_transform_features(dropped_test_df, n_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:05.523657Z","iopub.execute_input":"2025-01-01T01:27:05.523931Z","iopub.status.idle":"2025-01-01T01:27:06.045311Z","shell.execute_reply.started":"2025-01-01T01:27:05.523907Z","shell.execute_reply":"2025-01-01T01:27:06.044249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def sqr_transform_features(df, features):\n  \"\"\"\n  Applies np.log1p transformation to specified features in a Pandas DataFrame.\n\n  Args:\n    df: The Pandas DataFrame.\n    features: A list of strings, representing the column names to transform.\n\n  Returns:\n    A new Pandas DataFrame with the specified features transformed using np.log1p.\n  \"\"\"\n  df_copy = df.copy() # Create a copy to avoid modifying the original df\n  for feature in features:\n    if feature in df_copy.columns: # check to see if the feature is there\n      df_copy[feature] = np.sqrt(df_copy[feature])\n    else:\n      print(f\"Feature '{feature}' not found in DataFrame columns.\")\n  return df_copy\n\n#n_cols=['Annual Income', 'Health Score', 'Credit Score', 'sec_since_70' ]\n'''\nn_cols=['Health Score', 'Credit Score' ]\nX = sqr_transform_features(X, n_cols)\ndropped_test_df = sqr_transform_features(dropped_test_df, n_cols)\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:06.046866Z","iopub.execute_input":"2025-01-01T01:27:06.04718Z","iopub.status.idle":"2025-01-01T01:27:06.054589Z","shell.execute_reply.started":"2025-01-01T01:27:06.047153Z","shell.execute_reply":"2025-01-01T01:27:06.053602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:06.055731Z","iopub.execute_input":"2025-01-01T01:27:06.056056Z","iopub.status.idle":"2025-01-01T01:27:06.082572Z","shell.execute_reply.started":"2025-01-01T01:27:06.056005Z","shell.execute_reply":"2025-01-01T01:27:06.08159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:06.083793Z","iopub.execute_input":"2025-01-01T01:27:06.08413Z","iopub.status.idle":"2025-01-01T01:27:06.892645Z","shell.execute_reply.started":"2025-01-01T01:27:06.084102Z","shell.execute_reply":"2025-01-01T01:27:06.891577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#fill with mean median or just with a val -1\n#or give a diff neg val for missing in each feature?\ndef fill_nan_with_agg(df, columns_to_fill, aggregation_method='median'):\n    \"\"\"\n    Fills NaN values in specified columns of a Pandas DataFrame with either the median or mean or number say -1.\n\n    Args:\n        df: The Pandas DataFrame.\n        columns_to_fill: A list of column names to fill NaN values in.\n        aggregation_method:  'median' (default) or 'mean'.  Specifies how to fill NaN.\n\n    Returns:\n        A new Pandas DataFrame with NaN values filled. Returns None if input is invalid.\n    \"\"\"\n    if not isinstance(df, pd.DataFrame):\n        print(\"Error: Input must be a Pandas DataFrame.\")\n        return None\n    if not isinstance(columns_to_fill, list) or not all(isinstance(col, str) for col in columns_to_fill):\n        print(\"Error: columns_to_fill must be a list of strings.\")\n        return None\n    if not all(col in df.columns for col in columns_to_fill):\n        print(\"Error: Not all columns in columns_to_fill exist in the DataFrame.\")\n        return None\n    if aggregation_method not in ['median', 'mean', '-1']:\n        print(\"Error: aggregation_method must be 'median' or 'mean'.\")\n        return None\n\n\n    df_filled = df.copy()  # Create a copy to avoid modifying the original DataFrame\n\n    for col in columns_to_fill:\n        if df_filled[col].dtype in [np.float64, np.int64]: #Only fill numerical columns\n          if aggregation_method == 'median':\n              fill_value = df_filled[col].median()\n              df_filled[col] = df_filled[col].fillna(fill_value)\n          elif aggregation_method == 'mean':  # aggregation_method == 'mean'\n              fill_value = df_filled[col].mean()\n              df_filled[col] = df_filled[col].fillna(fill_value)\n          elif aggregation_method == '-1':  # aggregation_method == '-1'\n              #fill_value = df_filled[col].mean()\n              df_filled[col] = df_filled[col].fillna(-1)\n        else:\n          print(f\"Warning: Column '{col}' is not numeric; skipping NaN filling.\")\n\n\n    return df_filled\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:06.908397Z","iopub.execute_input":"2025-01-01T01:27:06.908725Z","iopub.status.idle":"2025-01-01T01:27:06.917477Z","shell.execute_reply.started":"2025-01-01T01:27:06.908698Z","shell.execute_reply":"2025-01-01T01:27:06.916438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ncolumns_to_fill = ['Annual Income', 'Health Score', 'Credit Score']\n\n# Fill with the -1\nX = fill_nan_with_agg(X, columns_to_fill, '-1')\n#print(\" X DataFrame filled with -1:\")\ndropped_test_df = fill_nan_with_agg(dropped_test_df, columns_to_fill, '-1')\n#print(\"test DataFrame filled with -1:\")\n\n'''\n# Fill with the median\nX = fill_nan_with_agg(X, columns_to_fill, 'median')\nprint(\" X DataFrame filled with median:\")\ndropped_test_df = fill_nan_with_agg(dropped_test_df, columns_to_fill, 'median')\nprint(\"test DataFrame filled with median:\")\n'''\n\n'''\n#Fill with the mean\ndf_filled_mean = fill_nan_with_agg(df, columns_to_fill, 'mean')\nprint(\"\\nDataFrame filled with mean:\")\nprint(df_filled_mean)\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:06.918925Z","iopub.execute_input":"2025-01-01T01:27:06.919365Z","iopub.status.idle":"2025-01-01T01:27:07.32861Z","shell.execute_reply.started":"2025-01-01T01:27:06.919328Z","shell.execute_reply":"2025-01-01T01:27:07.327398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#create encoding based on unique entries  freq\n#leave nans for next step\n#should have just made this as 'include columns'\ndef list_unique_entries(df, exclude_columns=None):\n    \"\"\"\n    Returns a list of lists, where each inner list contains unique entries for a column \n    in the dataframe, sorted by frequency in descending order.\n\n    Args:\n        df: The input Pandas DataFrame.\n\n    Returns:\n        A list of lists, where each inner list contains unique values for a column. Returns an empty list if the input dataframe is empty.\n    \"\"\"\n    if df.empty:\n        return []\n\n    unique_entry_lists = []  # Initialize the list to store results.\n    for col in df.columns:\n        if col not in exclude_columns:\n            unique_counts = df[col].value_counts()\n            unique_entries = unique_counts.index.tolist() # Extract unique values as a list\n            unique_entries = [entry for entry in unique_entries if not pd.isna(entry)] #exclude nans\n            unique_entry_lists.append(unique_entries)\n    return unique_entry_lists\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:07.329811Z","iopub.execute_input":"2025-01-01T01:27:07.330161Z","iopub.status.idle":"2025-01-01T01:27:07.336519Z","shell.execute_reply.started":"2025-01-01T01:27:07.33013Z","shell.execute_reply":"2025-01-01T01:27:07.335381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#X.head(30)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:07.337911Z","iopub.execute_input":"2025-01-01T01:27:07.338891Z","iopub.status.idle":"2025-01-01T01:27:07.346179Z","shell.execute_reply.started":"2025-01-01T01:27:07.338854Z","shell.execute_reply":"2025-01-01T01:27:07.345349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:07.347479Z","iopub.execute_input":"2025-01-01T01:27:07.347756Z","iopub.status.idle":"2025-01-01T01:27:07.905205Z","shell.execute_reply.started":"2025-01-01T01:27:07.347731Z","shell.execute_reply":"2025-01-01T01:27:07.904195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#stop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:07.906418Z","iopub.execute_input":"2025-01-01T01:27:07.90672Z","iopub.status.idle":"2025-01-01T01:27:07.910939Z","shell.execute_reply.started":"2025-01-01T01:27:07.906692Z","shell.execute_reply":"2025-01-01T01:27:07.909993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nexcl_cols =['Age', 'Number of Dependents','Previous Claims', 'Vehicle Age', 'Insurance Duration', \n            'Health Score','Credit Score','Annual Income', 'sec_since_70', '_year', '_month', \n           '_day', '_d_of_w', '_w_of_y']\nresult_list = list_unique_entries(X, excl_cols)\n'''\nfor item in result_list:\n    print(item)\n    print(\"------------------------\\n\\n\")\nprint('result_list len=  '+str(len(result_list)))\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:07.912302Z","iopub.execute_input":"2025-01-01T01:27:07.913225Z","iopub.status.idle":"2025-01-01T01:27:08.852235Z","shell.execute_reply.started":"2025-01-01T01:27:07.913185Z","shell.execute_reply":"2025-01-01T01:27:08.851334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result_list","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:08.853352Z","iopub.execute_input":"2025-01-01T01:27:08.853641Z","iopub.status.idle":"2025-01-01T01:27:08.860333Z","shell.execute_reply.started":"2025-01-01T01:27:08.853616Z","shell.execute_reply":"2025-01-01T01:27:08.859343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#basically making my own transform of unique values to a number based on frequ\n\ndef transform_list_of_lists(list_of_lists):\n\n    #Transforms a list of lists into a list of dictionaries.\n\n    result = []\n    for inner_list in list_of_lists:\n        dictionary = {}\n        count = 1\n        for item in inner_list:\n            dictionary[item] = count\n            count += 1\n        result.append(dictionary)\n    return result\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:08.861685Z","iopub.execute_input":"2025-01-01T01:27:08.862096Z","iopub.status.idle":"2025-01-01T01:27:08.871406Z","shell.execute_reply.started":"2025-01-01T01:27:08.862058Z","shell.execute_reply":"2025-01-01T01:27:08.870308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"list_of_dicts = transform_list_of_lists(result_list)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:08.872656Z","iopub.execute_input":"2025-01-01T01:27:08.87298Z","iopub.status.idle":"2025-01-01T01:27:08.883741Z","shell.execute_reply.started":"2025-01-01T01:27:08.872951Z","shell.execute_reply":"2025-01-01T01:27:08.882761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(list_of_dicts)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:08.884948Z","iopub.execute_input":"2025-01-01T01:27:08.885273Z","iopub.status.idle":"2025-01-01T01:27:08.897586Z","shell.execute_reply.started":"2025-01-01T01:27:08.885246Z","shell.execute_reply":"2025-01-01T01:27:08.896545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"list_of_dicts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:08.89885Z","iopub.execute_input":"2025-01-01T01:27:08.89919Z","iopub.status.idle":"2025-01-01T01:27:08.911873Z","shell.execute_reply.started":"2025-01-01T01:27:08.899162Z","shell.execute_reply":"2025-01-01T01:27:08.910855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_x_cols =dropped_test_df.columns.to_list()\nall_x_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:08.913436Z","iopub.execute_input":"2025-01-01T01:27:08.913873Z","iopub.status.idle":"2025-01-01T01:27:08.922894Z","shell.execute_reply.started":"2025-01-01T01:27:08.913831Z","shell.execute_reply":"2025-01-01T01:27:08.9218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"excl_cols\n#stop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:08.92453Z","iopub.execute_input":"2025-01-01T01:27:08.924822Z","iopub.status.idle":"2025-01-01T01:27:08.935321Z","shell.execute_reply.started":"2025-01-01T01:27:08.924795Z","shell.execute_reply":"2025-01-01T01:27:08.934442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"run_list = [\n     'Gender', \n     'Marital Status',     \n     'Education Level',\n     'Occupation', \n     'Location',\n     'Policy Type',        \n     'Customer Feedback',\n     'Smoking Status',\n     'Exercise Frequency',\n     'Property Type'\n           ]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:08.93641Z","iopub.execute_input":"2025-01-01T01:27:08.936705Z","iopub.status.idle":"2025-01-01T01:27:08.945809Z","shell.execute_reply.started":"2025-01-01T01:27:08.936679Z","shell.execute_reply":"2025-01-01T01:27:08.944962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"counter = 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:08.947171Z","iopub.execute_input":"2025-01-01T01:27:08.947867Z","iopub.status.idle":"2025-01-01T01:27:08.957167Z","shell.execute_reply.started":"2025-01-01T01:27:08.94783Z","shell.execute_reply":"2025-01-01T01:27:08.956211Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:08.958545Z","iopub.execute_input":"2025-01-01T01:27:08.958969Z","iopub.status.idle":"2025-01-01T01:27:08.987179Z","shell.execute_reply.started":"2025-01-01T01:27:08.958932Z","shell.execute_reply":"2025-01-01T01:27:08.986215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfor col in run_list:\n    X[col]=X[col].map(list_of_dicts[counter])\n    dropped_test_df[col]=dropped_test_df[col].map(list_of_dicts[counter]) \n    counter = counter+1\n\n\n################################################################","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:08.988525Z","iopub.execute_input":"2025-01-01T01:27:08.988932Z","iopub.status.idle":"2025-01-01T01:27:10.357164Z","shell.execute_reply.started":"2025-01-01T01:27:08.988895Z","shell.execute_reply":"2025-01-01T01:27:10.356087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:10.358681Z","iopub.execute_input":"2025-01-01T01:27:10.359112Z","iopub.status.idle":"2025-01-01T01:27:10.385351Z","shell.execute_reply.started":"2025-01-01T01:27:10.359071Z","shell.execute_reply":"2025-01-01T01:27:10.384292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dropped_test_df.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:10.386674Z","iopub.execute_input":"2025-01-01T01:27:10.386997Z","iopub.status.idle":"2025-01-01T01:27:10.415929Z","shell.execute_reply.started":"2025-01-01T01:27:10.386967Z","shell.execute_reply":"2025-01-01T01:27:10.414989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#stop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:10.41766Z","iopub.execute_input":"2025-01-01T01:27:10.418356Z","iopub.status.idle":"2025-01-01T01:27:10.422575Z","shell.execute_reply.started":"2025-01-01T01:27:10.418317Z","shell.execute_reply":"2025-01-01T01:27:10.421479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#fill cols with nans with a different integer for each column\ndef fill_nan_with_distinct_integers(df, columns_to_fill, fill_num):\n    \"\"\"\n    Fills NaN values in specified columns with distinct integers.\n\n    Args:\n        df: The Pandas DataFrame.\n        columns_to_fill: A list of column names to fill NaN values in.\n\n    Returns:\n        A new Pandas DataFrame with NaN values filled. Returns None if input is invalid.\n    \"\"\"\n\n    if not isinstance(df, pd.DataFrame):\n        print(\"Error: Input must be a Pandas DataFrame.\")\n        return None\n    if not isinstance(columns_to_fill, list) or not all(isinstance(col, str) for col in columns_to_fill):\n        print(\"Error: columns_to_fill must be a list of strings.\")\n        return None\n    if not all(col in df.columns for col in columns_to_fill):\n        print(\"Error: Not all columns in columns_to_fill exist in the DataFrame.\")\n        return None\n\n\n    df_filled = df.copy()\n    fill_values = {}  # Keep track of distinct fill values for each column\n    i=0\n    for col in columns_to_fill:\n        \n        if df_filled[col].dtype in [np.float64, np.int64]: #Only fill numerical columns\n\n            # Find a distinct integer value\n            #i = 0\n            while True:\n                fill_value = -fill_num - i  # Start with a negative number to avoid collisions with existing data\n                if fill_value not in df_filled[col].values:  #Check if the value is already present\n                  break\n                i += 1\n            fill_values[col] = fill_value #Save the value for this column\n            df_filled[col] = df_filled[col].fillna(fill_value)\n            i += 1\n        else:\n            print(f\"Warning: Column '{col}' is not numeric; skipping NaN filling.\")\n\n\n    return df_filled\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:10.42409Z","iopub.execute_input":"2025-01-01T01:27:10.424797Z","iopub.status.idle":"2025-01-01T01:27:10.436646Z","shell.execute_reply.started":"2025-01-01T01:27:10.424755Z","shell.execute_reply":"2025-01-01T01:27:10.435568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#fill cols with nans with a different integer for each column\ndef fill_nan_with_integer(df, columns_to_fill, fill_num):\n    \"\"\"\n    Fills NaN values in specified columns with one integers.\n\n    Args:\n        df: The Pandas DataFrame.\n        columns_to_fill: A list of column names to fill NaN values in.\n\n    Returns:\n        A new Pandas DataFrame with NaN values filled. Returns None if input is invalid.\n    \"\"\"\n\n    if not isinstance(df, pd.DataFrame):\n        print(\"Error: Input must be a Pandas DataFrame.\")\n        return None\n    if not isinstance(columns_to_fill, list) or not all(isinstance(col, str) for col in columns_to_fill):\n        print(\"Error: columns_to_fill must be a list of strings.\")\n        return None\n    if not all(col in df.columns for col in columns_to_fill):\n        print(\"Error: Not all columns in columns_to_fill exist in the DataFrame.\")\n        return None\n\n    \n    df_filled = df.copy()\n    #fill_values = {}  # Keep track of distinct fill values for each column\n    #i=0\n    for col in columns_to_fill:\n        \n        if df_filled[col].dtype in [np.float64, np.int64]: #Only fill numerical columns\n\n            df_filled[col] = df_filled[col].fillna(fill_num)\n\n        else:\n            print(f\"Warning: Column '{col}' is not numeric; skipping NaN filling.\")\n\n\n    return df_filled\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:10.437907Z","iopub.execute_input":"2025-01-01T01:27:10.438233Z","iopub.status.idle":"2025-01-01T01:27:10.452608Z","shell.execute_reply.started":"2025-01-01T01:27:10.438204Z","shell.execute_reply":"2025-01-01T01:27:10.451616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nAge                      12489\nMarital Status           12336\nNumber of Dependents     73130\nOccupation              239125\nPrevious Claims         242802\nCustomer Feedback        52276\n\nCredit Score             91451\nHealth Score             49449\nAnnual Income            29860\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:10.453859Z","iopub.execute_input":"2025-01-01T01:27:10.454179Z","iopub.status.idle":"2025-01-01T01:27:10.468385Z","shell.execute_reply.started":"2025-01-01T01:27:10.454152Z","shell.execute_reply":"2025-01-01T01:27:10.467338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns_to_fill = [     \n     'Marital Status',     \n     'Age',    \n     'Number of Dependents',    \n     'Occupation',     \n     'Previous Claims',\n     'Customer Feedback',    \n                ]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:10.469861Z","iopub.execute_input":"2025-01-01T01:27:10.470622Z","iopub.status.idle":"2025-01-01T01:27:10.479429Z","shell.execute_reply.started":"2025-01-01T01:27:10.47058Z","shell.execute_reply":"2025-01-01T01:27:10.478441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n#I have tries a sepearate cat for tarin and test - 100, 200 drtops down to 1.05 or so - \n#bad note traking but I think it was some of the earlier  lb score\nX = fill_nan_with_distinct_integers(X, columns_to_fill, 100)\nprint(\"X DataFrame with NaN values filled with distinct integers:\")\ndropped_test_df = fill_nan_with_distinct_integers(dropped_test_df, columns_to_fill, 100)\nprint(\"dropped_test_df DataFrame with NaN values filled with distinct integers:\")\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:10.480766Z","iopub.execute_input":"2025-01-01T01:27:10.481121Z","iopub.status.idle":"2025-01-01T01:27:10.491389Z","shell.execute_reply.started":"2025-01-01T01:27:10.481061Z","shell.execute_reply":"2025-01-01T01:27:10.49035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nX = fill_nan_with_integer(X, columns_to_fill, -1)\nprint(\"X DataFrame with NaN values filled with distinct integers:\")\ndropped_test_df = fill_nan_with_integer(dropped_test_df, columns_to_fill, -1)\nprint(\"dropped_test_df DataFrame with NaN values filled with distinct integers:\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:10.492767Z","iopub.execute_input":"2025-01-01T01:27:10.493107Z","iopub.status.idle":"2025-01-01T01:27:10.985971Z","shell.execute_reply.started":"2025-01-01T01:27:10.493077Z","shell.execute_reply":"2025-01-01T01:27:10.984987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:10.987254Z","iopub.execute_input":"2025-01-01T01:27:10.987569Z","iopub.status.idle":"2025-01-01T01:27:11.012822Z","shell.execute_reply.started":"2025-01-01T01:27:10.987541Z","shell.execute_reply":"2025-01-01T01:27:11.011686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#stop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.014076Z","iopub.execute_input":"2025-01-01T01:27:11.014405Z","iopub.status.idle":"2025-01-01T01:27:11.025619Z","shell.execute_reply.started":"2025-01-01T01:27:11.014373Z","shell.execute_reply":"2025-01-01T01:27:11.024499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nan_counts_X = X.isna().sum() \nnan_counts_X","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.026994Z","iopub.execute_input":"2025-01-01T01:27:11.027365Z","iopub.status.idle":"2025-01-01T01:27:11.0747Z","shell.execute_reply.started":"2025-01-01T01:27:11.027334Z","shell.execute_reply":"2025-01-01T01:27:11.073671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nan_counts_test = dropped_test_df.isna().sum()\nnan_counts_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.075806Z","iopub.execute_input":"2025-01-01T01:27:11.07612Z","iopub.status.idle":"2025-01-01T01:27:11.106217Z","shell.execute_reply.started":"2025-01-01T01:27:11.076095Z","shell.execute_reply":"2025-01-01T01:27:11.105211Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.107617Z","iopub.execute_input":"2025-01-01T01:27:11.108282Z","iopub.status.idle":"2025-01-01T01:27:11.16629Z","shell.execute_reply.started":"2025-01-01T01:27:11.108242Z","shell.execute_reply":"2025-01-01T01:27:11.165269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfloat_cols_to_int =[     \n     'Marital Status',     \n     'Age',    \n     'Number of Dependents',    \n     'Occupation',     \n     'Previous Claims',\n     'Customer Feedback',    \n     'Vehicle Age',\n     'Insurance Duration',\n     '_year',\n     '_month',\n     '_day',\n     '_d_of_w',\n     #'_w_of_y',\n    ]\n#if I want to do this conversion I have to fill na , pick 50 since its \nX[float_cols_to_int] = X[float_cols_to_int].astype(int)\ndropped_test_df[float_cols_to_int] = dropped_test_df[float_cols_to_int].astype(int) #astype('Int64') presereves the nans\n\n#X[float_cols_to_int] = X[float_cols_to_int].astype(str)\n#dropped_test_df[float_cols_to_int] = dropped_test_df[float_cols_to_int].astype(str) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.167635Z","iopub.execute_input":"2025-01-01T01:27:11.168312Z","iopub.status.idle":"2025-01-01T01:27:11.401199Z","shell.execute_reply.started":"2025-01-01T01:27:11.168271Z","shell.execute_reply":"2025-01-01T01:27:11.400383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#X.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.402357Z","iopub.execute_input":"2025-01-01T01:27:11.402675Z","iopub.status.idle":"2025-01-01T01:27:11.406889Z","shell.execute_reply.started":"2025-01-01T01:27:11.402646Z","shell.execute_reply":"2025-01-01T01:27:11.405849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.408089Z","iopub.execute_input":"2025-01-01T01:27:11.40838Z","iopub.status.idle":"2025-01-01T01:27:11.456057Z","shell.execute_reply.started":"2025-01-01T01:27:11.408353Z","shell.execute_reply":"2025-01-01T01:27:11.455092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#catcols is everything butincome, health score credit score, sec since 70","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.45732Z","iopub.execute_input":"2025-01-01T01:27:11.457714Z","iopub.status.idle":"2025-01-01T01:27:11.46211Z","shell.execute_reply.started":"2025-01-01T01:27:11.457677Z","shell.execute_reply":"2025-01-01T01:27:11.461085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_int_columns(df):\n  \"\"\"\n  Returns a list of column names from a Pandas DataFrame that have an integer data type.\n\n  Args:\n    df: The Pandas DataFrame.\n\n  Returns:\n    A list of strings, representing column names with integer data type.\n  \"\"\"\n\n  int_columns = []\n  for col in df.columns:\n     if pd.api.types.is_integer_dtype(df[col]):\n        int_columns.append(col)\n  return int_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.463425Z","iopub.execute_input":"2025-01-01T01:27:11.463794Z","iopub.status.idle":"2025-01-01T01:27:11.471521Z","shell.execute_reply.started":"2025-01-01T01:27:11.463758Z","shell.execute_reply":"2025-01-01T01:27:11.470509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"int_cols = get_int_columns(X)\nint_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.472711Z","iopub.execute_input":"2025-01-01T01:27:11.473108Z","iopub.status.idle":"2025-01-01T01:27:11.485745Z","shell.execute_reply.started":"2025-01-01T01:27:11.473078Z","shell.execute_reply":"2025-01-01T01:27:11.48487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#stop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.486838Z","iopub.execute_input":"2025-01-01T01:27:11.487154Z","iopub.status.idle":"2025-01-01T01:27:11.495256Z","shell.execute_reply.started":"2025-01-01T01:27:11.487127Z","shell.execute_reply":"2025-01-01T01:27:11.494323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#num_cols= X.select_dtypes(include='number').columns.tolist()\n#num_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.496724Z","iopub.execute_input":"2025-01-01T01:27:11.497068Z","iopub.status.idle":"2025-01-01T01:27:11.504393Z","shell.execute_reply.started":"2025-01-01T01:27:11.497013Z","shell.execute_reply":"2025-01-01T01:27:11.503569Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\ndef one_hot_encode_train_test(train_df, test_df, categorical_columns, fill_nan_strategy=None):\n    \"\"\"\n    One-hot encodes specified categorical columns in both training and testing DataFrames,\n    ensuring consistent encoding.\n\n    Args:\n        train_df: The training Pandas DataFrame.\n        test_df: The testing Pandas DataFrame.\n        categorical_columns: A list of column names (strings) to one-hot encode.\n        fill_nan_strategy: If 'fill_with_string', missing values will be filled with the string 'Missing'.\n                         If None, nan values are handled using the handle_unknown='ignore' parameter\n                         Defaults to None.\n\n    Returns:\n        A tuple containing:\n          - The transformed training DataFrame.\n          - The transformed testing DataFrame.\n    \"\"\"\n    train_df_copy = train_df.copy()\n    test_df_copy = test_df.copy()\n\n    # Fill nan values before fitting the one hot encoder, if fill_nan_strategy is set\n    if fill_nan_strategy == 'fill_with_string':\n        for col in categorical_columns:\n            train_df_copy[col] = train_df_copy[col].fillna('Missing')\n            test_df_copy[col] = test_df_copy[col].fillna('Missing')\n\n    # 1. Initialize OneHotEncoder with handle_unknown = 'ignore'\n    ohe = OneHotEncoder(handle_unknown='ignore', sparse_output=False)\n\n    # 2. Fit the OneHotEncoder on the training data\n    ohe.fit(train_df_copy[categorical_columns])\n\n    # 3. Transform both the train and test DataFrames\n    train_encoded = ohe.transform(train_df_copy[categorical_columns])\n    test_encoded = ohe.transform(test_df_copy[categorical_columns])\n\n     # create dataframes and add the encoded data to the original dataframes, as well as removing original columns.\n    train_encoded_df = pd.DataFrame(train_encoded, columns = ohe.get_feature_names_out(categorical_columns))\n    test_encoded_df = pd.DataFrame(test_encoded, columns = ohe.get_feature_names_out(categorical_columns))\n\n    train_df_encoded = pd.concat([train_df_copy, train_encoded_df], axis=1).drop(columns=categorical_columns)\n    test_df_encoded = pd.concat([test_df_copy, test_encoded_df], axis=1).drop(columns=categorical_columns)\n\n    return train_df_encoded, test_df_encoded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.505977Z","iopub.execute_input":"2025-01-01T01:27:11.506539Z","iopub.status.idle":"2025-01-01T01:27:11.515806Z","shell.execute_reply.started":"2025-01-01T01:27:11.506501Z","shell.execute_reply":"2025-01-01T01:27:11.514796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#categorical_cols1 = ['int_categorical_with_nan', 'another_categorical']\n####THIS DOES NO BETTERE THAN M Y MAKING ALL CATS AND SPECIFYING CAT FEATURES FOR MODELS\n\n\n#categorical_cols1 = int_cols\n#X, dropped_test_df = one_hot_encode_train_test(X, dropped_test_df, categorical_cols1)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.517073Z","iopub.execute_input":"2025-01-01T01:27:11.517762Z","iopub.status.idle":"2025-01-01T01:27:11.531587Z","shell.execute_reply.started":"2025-01-01T01:27:11.517714Z","shell.execute_reply":"2025-01-01T01:27:11.53063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\ndef draw_plots(the_cols, the_outcome, the_dataset):\n    for col in the_cols:\n\n        ax=sns.violinplot(data=the_dataset, x=the_outcome, y=col, palette='turbo', inner=None, linewidth=0, saturation=0.4)\n        sns.boxplot(data=the_dataset,x=the_outcome, y=col, palette='turbo', width=0.3, boxprops={'zorder':2}, ax=ax, saturation=0.6)\n        \n\n        plt.show()\n ''' ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.53282Z","iopub.execute_input":"2025-01-01T01:27:11.53317Z","iopub.status.idle":"2025-01-01T01:27:11.541288Z","shell.execute_reply.started":"2025-01-01T01:27:11.53314Z","shell.execute_reply":"2025-01-01T01:27:11.540461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#draw_plots(num_cols, TARGET, dropped_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.542633Z","iopub.execute_input":"2025-01-01T01:27:11.54293Z","iopub.status.idle":"2025-01-01T01:27:11.550825Z","shell.execute_reply.started":"2025-01-01T01:27:11.542903Z","shell.execute_reply":"2025-01-01T01:27:11.549805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#could try pairplots with specific features??\n#needs more mem than I have acces to\n'''\nsns.set_style('whitegrid');\nsns.pairplot(dropped_df, hue=TARGET, height=3);\nplt.show()\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.552095Z","iopub.execute_input":"2025-01-01T01:27:11.552513Z","iopub.status.idle":"2025-01-01T01:27:11.561247Z","shell.execute_reply.started":"2025-01-01T01:27:11.552485Z","shell.execute_reply":"2025-01-01T01:27:11.56018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.562477Z","iopub.execute_input":"2025-01-01T01:27:11.562806Z","iopub.status.idle":"2025-01-01T01:27:11.689857Z","shell.execute_reply.started":"2025-01-01T01:27:11.562773Z","shell.execute_reply":"2025-01-01T01:27:11.688779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#X.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.691089Z","iopub.execute_input":"2025-01-01T01:27:11.691418Z","iopub.status.idle":"2025-01-01T01:27:11.699991Z","shell.execute_reply.started":"2025-01-01T01:27:11.691385Z","shell.execute_reply":"2025-01-01T01:27:11.699069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#dropped_test_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.70127Z","iopub.execute_input":"2025-01-01T01:27:11.701578Z","iopub.status.idle":"2025-01-01T01:27:11.709902Z","shell.execute_reply.started":"2025-01-01T01:27:11.701549Z","shell.execute_reply":"2025-01-01T01:27:11.70881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nn_cols=['Annual Income', 'Health Score', 'Credit Score', 'sec_since_70' ]\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import StandardScaler\n\n\n#mms = MinMaxScaler(feature_range=(0, 1))\nmms = StandardScaler()\n# Scaling secific columns\nX[n_cols] = mms.fit_transform(X[n_cols])\ndropped_test_df[n_cols] = mms.transform(dropped_test_df[n_cols])\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.711394Z","iopub.execute_input":"2025-01-01T01:27:11.712154Z","iopub.status.idle":"2025-01-01T01:27:11.723773Z","shell.execute_reply.started":"2025-01-01T01:27:11.712113Z","shell.execute_reply":"2025-01-01T01:27:11.722857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#or other scalers\n\n#n_cols=['Annual Income', 'Health Score', 'Credit Score', 'sec_since_70' ]\n'''\nfrom sklearn.preprocessing import QuantileTransformer\nfrom sklearn.preprocessing import PowerTransformer\nfrom sklearn.preprocessing import RobustScaler\n#pt = RobustScaler(with_centering=False, with_scaling=True) #with_centering=True centers at zero and in this case I have others that are 0 or a +ve\n#pt = QuantileTransformer(output_distribution='normal')\npt = PowerTransformer(method='yeo-johnson') #or method='box_cox' if all valeus in features are > 0, if not yeo-johnson\nX[n_cols]= pt.fit_transform(X[n_cols])\ndropped_test_df[n_cols] = pt.transform(dropped_test_df[n_cols]) \n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.725215Z","iopub.execute_input":"2025-01-01T01:27:11.725906Z","iopub.status.idle":"2025-01-01T01:27:11.738528Z","shell.execute_reply.started":"2025-01-01T01:27:11.725863Z","shell.execute_reply":"2025-01-01T01:27:11.73767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n#or just do:\n#Im doing before filling missing\ndef log1p_transform_features(df, features):\n  \"\"\"\n  Applies np.log1p transformation to specified features in a Pandas DataFrame.\n\n  Args:\n    df: The Pandas DataFrame.\n    features: A list of strings, representing the column names to transform.\n\n  Returns:\n    A new Pandas DataFrame with the specified features transformed using np.log1p.\n  \"\"\"\n  df_copy = df.copy() # Create a copy to avoid modifying the original df\n  for feature in features:\n    if feature in df_copy.columns: # check to see if the feature is there\n      df_copy[feature] = np.log1p(df_copy[feature])\n    else:\n      print(f\"Feature '{feature}' not found in DataFrame columns.\")\n  return df_copy\n\nX = log1p_transform_features(X, n_cols)\ndropped_test_df = log1p_transform_features(dropped_test_df, n_cols)\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.739942Z","iopub.execute_input":"2025-01-01T01:27:11.740388Z","iopub.status.idle":"2025-01-01T01:27:11.750944Z","shell.execute_reply.started":"2025-01-01T01:27:11.740348Z","shell.execute_reply":"2025-01-01T01:27:11.75Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#X.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.752378Z","iopub.execute_input":"2025-01-01T01:27:11.752796Z","iopub.status.idle":"2025-01-01T01:27:11.762068Z","shell.execute_reply.started":"2025-01-01T01:27:11.752758Z","shell.execute_reply":"2025-01-01T01:27:11.761138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"int_cols = X.select_dtypes(include='int').columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.763365Z","iopub.execute_input":"2025-01-01T01:27:11.763769Z","iopub.status.idle":"2025-01-01T01:27:11.989421Z","shell.execute_reply.started":"2025-01-01T01:27:11.763729Z","shell.execute_reply":"2025-01-01T01:27:11.988308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"int_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:11.990514Z","iopub.execute_input":"2025-01-01T01:27:11.99085Z","iopub.status.idle":"2025-01-01T01:27:12.000535Z","shell.execute_reply.started":"2025-01-01T01:27:11.990821Z","shell.execute_reply":"2025-01-01T01:27:11.999179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#In this case Im making all cols cats  and just using the cat hyp p in  xgb, cat and lgbc\ncategorical_features_indices = np.where(X.dtypes == int)[0]\ncategorical_features_indices\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:12.001991Z","iopub.execute_input":"2025-01-01T01:27:12.002442Z","iopub.status.idle":"2025-01-01T01:27:12.011642Z","shell.execute_reply.started":"2025-01-01T01:27:12.002399Z","shell.execute_reply":"2025-01-01T01:27:12.010311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#X.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:12.013133Z","iopub.execute_input":"2025-01-01T01:27:12.013542Z","iopub.status.idle":"2025-01-01T01:27:12.022856Z","shell.execute_reply.started":"2025-01-01T01:27:12.013502Z","shell.execute_reply":"2025-01-01T01:27:12.021946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#y_log = np.log1p(Y)\n#Y= np.log1p(Y)\n#Then later:\n#y_pred_original = np.expm1(y_pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:12.0241Z","iopub.execute_input":"2025-01-01T01:27:12.024528Z","iopub.status.idle":"2025-01-01T01:27:12.032811Z","shell.execute_reply.started":"2025-01-01T01:27:12.024489Z","shell.execute_reply":"2025-01-01T01:27:12.031762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:12.03446Z","iopub.execute_input":"2025-01-01T01:27:12.035123Z","iopub.status.idle":"2025-01-01T01:27:13.133885Z","shell.execute_reply.started":"2025-01-01T01:27:12.035082Z","shell.execute_reply":"2025-01-01T01:27:13.132906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.1351Z","iopub.execute_input":"2025-01-01T01:27:13.135416Z","iopub.status.idle":"2025-01-01T01:27:13.177359Z","shell.execute_reply.started":"2025-01-01T01:27:13.135388Z","shell.execute_reply":"2025-01-01T01:27:13.17635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Y.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.178547Z","iopub.execute_input":"2025-01-01T01:27:13.178849Z","iopub.status.idle":"2025-01-01T01:27:13.193439Z","shell.execute_reply.started":"2025-01-01T01:27:13.178821Z","shell.execute_reply":"2025-01-01T01:27:13.192598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dropped_test_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.194534Z","iopub.execute_input":"2025-01-01T01:27:13.194825Z","iopub.status.idle":"2025-01-01T01:27:13.227307Z","shell.execute_reply.started":"2025-01-01T01:27:13.194799Z","shell.execute_reply":"2025-01-01T01:27:13.226348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#stop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.228428Z","iopub.execute_input":"2025-01-01T01:27:13.228722Z","iopub.status.idle":"2025-01-01T01:27:13.232848Z","shell.execute_reply.started":"2025-01-01T01:27:13.228696Z","shell.execute_reply":"2025-01-01T01:27:13.231888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Im using CV method so I dont really need this do I, ecept that this maybe using a diff method to cv to pull ot the test cases so why not\n#X_train, X_test, y_train, y_test = train_test_split(X, Y,test_size=0.25, stratify=Y,random_state=42)\nX_train, X_test, y_train, y_test = train_test_split(X, Y,test_size=0.25, random_state=42)\n#or should I tranform Y here? so y_train gets transformed and not y_test and then when I run my predictions I tranform back beofre my rmsle score??","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.234073Z","iopub.execute_input":"2025-01-01T01:27:13.234385Z","iopub.status.idle":"2025-01-01T01:27:13.793529Z","shell.execute_reply.started":"2025-01-01T01:27:13.234351Z","shell.execute_reply":"2025-01-01T01:27:13.792697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ny_train = np.log1p(y_train)\ny_test = np.log1p(y_test) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.794823Z","iopub.execute_input":"2025-01-01T01:27:13.79551Z","iopub.status.idle":"2025-01-01T01:27:13.823616Z","shell.execute_reply.started":"2025-01-01T01:27:13.795471Z","shell.execute_reply":"2025-01-01T01:27:13.822832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#defined for all cases run\n#maybe its worth trying a diff cv split for the diff models??\nfrom sklearn.model_selection import RepeatedStratifiedKFold, RepeatedKFold, KFold\n'''\ncv_method = RepeatedStratifiedKFold(\n            n_splits=8,\n            n_repeats=2,\n            random_state=42)\n'''\n'''\ncv_method = KFold(\n            n_splits=8, \n            shuffle=True,\n            random_state=42)\n'''\n\ncv_method = RepeatedKFold(\n            n_splits=8, \n            n_repeats=1,\n            random_state=42)\n\nall_results_dict = {}\nimport pickle\n\npickle_path=\"/kaggle/working/\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.824871Z","iopub.execute_input":"2025-01-01T01:27:13.825191Z","iopub.status.idle":"2025-01-01T01:27:13.830309Z","shell.execute_reply.started":"2025-01-01T01:27:13.825163Z","shell.execute_reply":"2025-01-01T01:27:13.829349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#used this to begin with to help read/formnat what I saw, then just pulled from the actual cv scores\ndef print_grid_score(iters,the_dict, grid_name, which_score):\n    count = iters\n    \n    while count != 0:\n\n        print(f\"Coordinate descent {grid_name+str(count)} ACC score: {np.round(the_dict[grid_name+str(count)][which_score],4)}\")\n        count = count -1\n        \n#MY_SCORE='test roc auc score'\n\nMY_SCORE='root_mean_squared_log_error'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.831566Z","iopub.execute_input":"2025-01-01T01:27:13.831939Z","iopub.status.idle":"2025-01-01T01:27:13.839255Z","shell.execute_reply.started":"2025-01-01T01:27:13.831904Z","shell.execute_reply":"2025-01-01T01:27:13.838279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#from sklearn.metrics import get_scorer_names\n#all_scorers = get_scorer_names()\n#all_scorers","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.840565Z","iopub.execute_input":"2025-01-01T01:27:13.841093Z","iopub.status.idle":"2025-01-01T01:27:13.848074Z","shell.execute_reply.started":"2025-01-01T01:27:13.841053Z","shell.execute_reply":"2025-01-01T01:27:13.84733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Whats my metric? Just change it here\nfrom sklearn.metrics import roc_auc_score, make_scorer, matthews_corrcoef, accuracy_score \nfrom sklearn.metrics import  root_mean_squared_log_error, mean_absolute_error ,mean_squared_log_error, root_mean_squared_error\n#my_f1 = make_scorer(f1_score , average='weighted')\n#my_f1 = make_scorer(roc_auc_score , average='weighted')\n#my_f1 = 'roc_auc'\n\n#my_refit = 'roc_auc'\n#my_f1 = 'roc_auc_ovr_weighted'\n#my_f1 = make_scorer(matthews_corrcoef)\n#my_f1 = 'matthews_corrcoef'\n#my_f1= 'root_mean_squared_log_error'\n#my_f1='mean_squared_log_error'\n\n'''\n#can be written as:? check and if yes  make the change?\ndef rmsle_scorer(y_true, y_pred):\n    return -np.sqrt(mean_squared_log_error(y_true, y_pred))\n\nmy_f1=make_scorer(rmsle_scorer)\n'''\n#or try:\n'''\n# from: https://www.kaggle.com/code/cdeotte/metric-rsmle-vs-rsme-vs-mse-vs-mae\ndef rmsle_score(y_true, y_pred):\n    true_log = np.log1p(y_true)\n    pred_log = np.log1p(y_pred)\n    m = np.sqrt(np.mean( (true_log-pred_log)**2.0 ))\n    return m\n '''   \n\n\ndef rmsle_score(y_true, y_pred):\n    #Custom RMSLE metric \n    y_true = np.expm1(y_true) # undo log transformation of the target\n    y_pred = np.expm1(y_pred) # undo log transformation of the predictions\n    #return np.sqrt(mean_squared_log_error(y_true, y_pred))\n    return root_mean_squared_log_error(y_true, y_pred)\n\n'''\ndef rmsle_score(y_true, y_pred):\n    #Custom RMSLE metric \n    y_true = y_true \n    y_pred = y_pred \n    return np.sqrt(mean_squared_log_error(y_true, y_pred))\n'''\n'''\ndef rmsle_score(y_true, y_pred):\n    #Custom RMSLE metric \n    #return np.sqrt(mean_squared_log_error(y_true, y_pred))\n    return root_mean_squared_log_error(y_true, y_pred)\n'''\n# Create a scorer object using make_scorer (Corrected: explicitly stating less_is_better)\nrmsle_scorer = make_scorer(rmsle_score, response_method='predict', greater_is_better=False) # lower RMSLE is better\n#rmsle_scorer = 'neg_root_mean_squared_log_error'\n\n\n#def rmsle_score(y_true, y_pred):\n#    \"\"\"Custom RMSLE metric for CatBoost.\"\"\"\n#    return np.sqrt(mean_squared_log_error(y_true, y_pred))\n\n\n#funny if I reference the make scoer object and then pass that to my genberic functions it ode snot work?\n#in this case if i do scoring = my_f1 my function fails\n#But if I do scoring=rsmle_scorer I succeed Why?\nmy_scorer = rmsle_scorer","metadata":{"_kg_hide-input":true,"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.849348Z","iopub.execute_input":"2025-01-01T01:27:13.849655Z","iopub.status.idle":"2025-01-01T01:27:13.86477Z","shell.execute_reply.started":"2025-01-01T01:27:13.84962Z","shell.execute_reply":"2025-01-01T01:27:13.863863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#stop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.866192Z","iopub.execute_input":"2025-01-01T01:27:13.867063Z","iopub.status.idle":"2025-01-01T01:27:13.879364Z","shell.execute_reply.started":"2025-01-01T01:27:13.866994Z","shell.execute_reply":"2025-01-01T01:27:13.878439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#extracting default parameters from benchmark model\n#superflouos sinnce all models give this, but this at least is generalized to deal with differences with catboost\ndef get_default_params(model, cat_flag=False):\n    default_params = {}\n    if(cat_flag):\n        gparams = model.get_all_params()\n    else:\n        gparams = model.get_params()\n    #print(gparams)\n    #default parameters have to be wrapped in lists - even single values - so GridSearchCV can take them as inputs ?\n    \n    for key in gparams.keys():\n        gp = gparams[key]\n        default_params[key] = [gp]\n    #print(default_params) \n\n    return default_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.880691Z","iopub.execute_input":"2025-01-01T01:27:13.881121Z","iopub.status.idle":"2025-01-01T01:27:13.892218Z","shell.execute_reply.started":"2025-01-01T01:27:13.881082Z","shell.execute_reply":"2025-01-01T01:27:13.891334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#def rmse(score):\n#   rmse = np.sqrt(-score) #do I log this now?\n#    return rmse\n    \n    #np.log1p","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.89348Z","iopub.execute_input":"2025-01-01T01:27:13.893832Z","iopub.status.idle":"2025-01-01T01:27:13.901344Z","shell.execute_reply.started":"2025-01-01T01:27:13.893804Z","shell.execute_reply":"2025-01-01T01:27:13.900379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n#change this for m ore generality instead of catFlag just a flag for cat, lightgbm or xgbm\ndef run_base_model(the_model, params=False, cat_flag=False):\n    #use sklearn for a cv_score on the base model\n    #print(f\"Model passed into function is of type: {type(the_model)}\") # Check type here\n    cross_val_scores = cross_val_score(the_model,X_train,y_train, scoring=my_scorer, cv=cv_method)\n    \n    df = pd.DataFrame(cross_val_scores)\n\n    print(\"Cross-validation scores:\", cross_val_scores)\n    print(\"Mean CV score:\", cross_val_scores.mean())\n\n# Now for making predictions and calculating the metric on test set we will need to inverse transform predictions\n    the_model.fit(X_train, y_train)\n    \n    y_pred_transformed = the_model.predict(X_test)\n    #y_pred = np.expm1(y_pred_transformed)\n    \n    #r_msle = root_mean_squared_log_error(np.expm1(y_test_trans), y_pred)\n    #r_msle = rmsle_score(np.expm1(y_test_trans), y_pred) #so based on this I actually dont need to transform y_test?\n    r_msle = rmsle_score(y_test, y_pred_transformed)\n    print(f'The test RMSLE score: {r_msle}')\n    \n    std_cv_score = cross_val_scores.std()\n    mean_cv_score = cross_val_scores.mean()    \n    #test_predictions = the_model.fit(X_train, y_train, verbose=0).predict(X_test)\n    #r_msle = rmsle_score(y_test, test_predictions)\n\n    if (cat_flag):\n        bp= the_model.get_all_params()\n    else:\n        bp = the_model.get_params()\n    result_dict = {'classifier': deepcopy(the_model),\n                         'cv_results': df.copy(),\n\n                         #'cfm_test': cfm_test,\n                         'mean_cv': mean_cv_score,\n                         'std_cv': std_cv_score,\n                         'rmsle': r_msle,\n                         \n                         'best_params': bp}\n    return result_dict,bp\n\n\n#implement above to below","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.902587Z","iopub.execute_input":"2025-01-01T01:27:13.902889Z","iopub.status.idle":"2025-01-01T01:27:13.911663Z","shell.execute_reply.started":"2025-01-01T01:27:13.902863Z","shell.execute_reply":"2025-01-01T01:27:13.91076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#random_states = [42, 123, 456, 789, 1011, 1213]\ndef run_base_model_many_splits(the_model, X, y, test_size, random_states):\n\n    results = []\n    best_rmsle = float('inf')  # Initialize with a large value\n    best_random_state = None\n\n    for random_state in random_states:\n        X_train_i, X_test_i, y_train_t, y_test_t = train_test_split(\n            X, y, test_size=test_size, random_state=random_state\n        )\n        #y_train_t = np.log1p(y_train_i)\n        #y_test_t = np.log1p(y_test_i) \n\n        model = clone(the_model) #Creating a copy of the model for each iteration\n\n        cross_val_scores = cross_val_score(model, X_train_i, y_train_t, scoring=rmsle_scorer, cv=cv_method)\n        df = pd.DataFrame(cross_val_scores)\n###################################################\n\n        model.fit(X_train_i, y_train_t)\n        y_pred_transformed = the_model.predict(X_test_i)\n        #y_pred = np.expm1(y_pred_transformed)\n        r_msle = rmsle_score(y_test_t, y_pred_transformed)\n        print(f'The test RMSLE score: {r_msle}')\n######################################################      \n        std_cv_score = cross_val_scores.std()\n        mean_cv_score = cross_val_scores.mean()\n        #test_predictions = model.fit(X_train, y_train, verbose=0).predict(X_test)\n        #r_msle = rmsle_score(y_test, test_predictions) #Use scorer function here\n\n        result_dict = {\n            'classifier': deepcopy(model),\n            'cv_results': df.copy(),\n            'mean_cv': mean_cv_score,\n            'std_cv': std_cv_score,\n            'rmsle': r_msle,\n            'best_params': model.get_params(),\n            'random_state': random_state\n        }\n        if r_msle < best_rmsle:\n            best_rmsle = r_msle\n            best_random_state = random_state\n        results.append(result_dict)\n        \n\n    return results, best_rmsle, best_random_state, ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.912904Z","iopub.execute_input":"2025-01-01T01:27:13.913254Z","iopub.status.idle":"2025-01-01T01:27:13.926974Z","shell.execute_reply.started":"2025-01-01T01:27:13.913227Z","shell.execute_reply":"2025-01-01T01:27:13.926086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef get_grid_cv(params_grid, run_model, model_name, cat_flag=False):\n    \n        result_dict = {}\n        count = 0\n      \n        this_grid_clf = GridSearchCV(estimator=run_model, scoring=rmsle_scorer, refit=True, param_grid=params_grid, verbose=1, cv=cv_method)\n\n        this_grid_clf.fit(X_train, y_train)\n        df = pd.DataFrame(this_grid_clf.cv_results_)\n###########################################################################\n        best_model = this_grid_clf.best_estimator_\n\n# Make predictions on the test set (remember to inverse transform if needed)\n        y_pred_transformed = best_model.predict(X_test)\n        #y_pred = np.expm1(y_pred_transformed)\n        \n\n        #r_msle = root_mean_squared_log_error(np.expm1(y_test), y_pred)\n        #r_msle = root_mean_squared_log_error(np.expm1(y_test), y_pred)\n        r_msle = rmsle_score(y_test, y_pred_transformed)\n        print(f'The test RMSLE score: {r_msle}')\n\n# Print results\n        print(\"Best parameters:\", this_grid_clf.best_params_)\n        print(\"Best cross-validation score:\", this_grid_clf.best_score_)\n###############################################################################\n\n        #best_model = this_grid_clf.best_estimator_\n\n        #test_score = best_model.score(X_test, y_test)\n        #print(f\"Test score: {test_score}\")\n    \n        #test_predictions = this_grid_clf.predict(X_test)\n\n        #orig_sized_test_preds = np.expm1(test_predictions) \n\n        #r_msle = rmsle_score(y_test, orig_sized_test_preds)\n\n        bs = this_grid_clf.best_score_\n        bp = this_grid_clf.best_params_\n        \n        with open(pickle_path+model_name+'_'+str(count+1)+'.pickle', 'wb') as to_write:\n            pickle.dump(this_grid_clf, to_write)\n        #count = count + 1\n        result_dict[f'{model_name}_{count+1}'] = {\n                                    'classifier': deepcopy(this_grid_clf), #probably dont need this since Im explicitly returning this below\n                                    'cv_results': df.copy(),\n\n                                    #'mean_cv': mean_cv_score,\n                                    'best_score': bs,\n                                    'rmsle': r_msle,\n                                    'best_params': bp}\n\n        return result_dict , this_grid_clf\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.928168Z","iopub.execute_input":"2025-01-01T01:27:13.928457Z","iopub.status.idle":"2025-01-01T01:27:13.941452Z","shell.execute_reply.started":"2025-01-01T01:27:13.928431Z","shell.execute_reply":"2025-01-01T01:27:13.940521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_random_searchcv( params_grid, base_model, model_name, n_iter_factor=0.5):\n\n    #its = np.cumsum([len(x) for x in params_grid.values()])[-1] #what should n_iter be set to? the num of permutations of the param grid in a grid search??\n    #I guess the more combinations (n_iters), the longer it takes the more like grid search it becomes?\n    #try random divide by 3?\n    #its = its//2\n    its = int(np.sum([len(v) for v in params_grid.values()]) * n_iter_factor)\n    its = max(1, its) # Ensure at least one iteration\n    this_rs_clf = RandomizedSearchCV(estimator=base_model, param_distributions=params_grid, scoring=rmsle_scorer, refit=True, verbose=1, cv=cv_method, n_iter=its)#return_train_score=True, is calc intensive to be True\n    #this_rs_clf.fit(X_train, y_train.values.ravel()) already doing ravel on way in and since we do cv no need to the train or test since cv divides into groups based on cv  method\n    this_rs_clf.fit(X_train, y_train)#.values.ravel())\n    df = pd.DataFrame(this_rs_clf.cv_results_)\n##################################################################\n    best_model = this_rs_clf.best_estimator_\n\n# Make predictions on the test set (remember to inverse transform if needed)\n    y_pred_transformed = best_model.predict(X_test)\n    #y_pred = np.expm1(y_pred_transformed)\n\n    r_msle = rmsle_score(y_test, y_pred_transformed)\n    print(f'The test RMSLE score: {r_msle}')\n\n# Print results\n    print(\"Best parameters:\", this_rs_clf.best_params_)\n    print(\"Best cross-validation score:\", this_rs_clf.best_score_)\n####################################################################\n    \n    #best_model = this_rs_clf.best_estimator_\n    #test_predictions = best_model.predict(X_test)\n    #r_msle = rmsle_score(y_test, test_predictions)\n    \n    #test_score = best_model.score(X_test, y_test)\n    #print(f\"Test score: {test_score}\")\n\n    bs = this_rs_clf.best_score_\n    bp = this_rs_clf.best_params_\n    result_rs_dicts = {}\n    \n    with open(pickle_path+model_name+'.pickle', 'wb') as to_write:\n        pickle.dump(this_rs_clf, to_write)\n\n    result_rs_dicts[f'{model_name}_'] = {'classifier': deepcopy(this_rs_clf),\n                                'cv_results': df.copy(),\n\n                                #'mean_cv': mean_cv_score,\n                                'rmsle': r_msle,\n                                'best_params': bp,\n                                'best_score':bs,       \n                                }\n\n    return result_rs_dicts, bp, this_rs_clf #I dont need to return bp since its already in the return results dict\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.942924Z","iopub.execute_input":"2025-01-01T01:27:13.943799Z","iopub.status.idle":"2025-01-01T01:27:13.958309Z","shell.execute_reply.started":"2025-01-01T01:27:13.943761Z","shell.execute_reply":"2025-01-01T01:27:13.95734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#STOP\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.959804Z","iopub.execute_input":"2025-01-01T01:27:13.960355Z","iopub.status.idle":"2025-01-01T01:27:13.970785Z","shell.execute_reply.started":"2025-01-01T01:27:13.96031Z","shell.execute_reply":"2025-01-01T01:27:13.969864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#I havent spent much time on any of the grad boost models, I should\n#so im leaving the 4 following for that project","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.972274Z","iopub.execute_input":"2025-01-01T01:27:13.972961Z","iopub.status.idle":"2025-01-01T01:27:13.979781Z","shell.execute_reply.started":"2025-01-01T01:27:13.972928Z","shell.execute_reply":"2025-01-01T01:27:13.978805Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **DecisionTreeClassifier**","metadata":{}},{"cell_type":"code","source":"'''\n%%time\n#what are bnasics for dtc?\n\ndtc_base = DecisionTreeClassifier()\ndtc_base.fit(X_train , y_train)\ntest_ras = roc_auc_score(y_test, dtc_base.predict_proba(X_test)[:,1])\n\nprint(f\"Base AUC score: {np.round(test_ras,4)}\")    \n\n#what can be added?\ndtc0 = DecisionTreeClassifier(criterion='entropy',class_weight='balanced', random_state=42, )\ndtc0.fit(X_train , y_train)\n\ndtc0_params = get_default_params(dtc0)\nbase_dict_dtc = {}\n#remember changed this whole function, \nbase_dict_dtc['dtc0'], dtc0_best_params = run_base_model(dtc0, dtc0_params)\n\nwith open(pickle_path+'dtc_clf0.pickle', 'wb') as to_write:\n    pickle.dump(dtc0, to_write)\n '''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.980976Z","iopub.execute_input":"2025-01-01T01:27:13.981307Z","iopub.status.idle":"2025-01-01T01:27:13.990075Z","shell.execute_reply.started":"2025-01-01T01:27:13.981281Z","shell.execute_reply":"2025-01-01T01:27:13.989223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#print(f\"Benchmark AUC score: {np.round(base_dict_dtc['dtc0']['test roc auc score'],4)}\")  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.991408Z","iopub.execute_input":"2025-01-01T01:27:13.991727Z","iopub.status.idle":"2025-01-01T01:27:13.998421Z","shell.execute_reply.started":"2025-01-01T01:27:13.9917Z","shell.execute_reply":"2025-01-01T01:27:13.99761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#dtc_base_params = dtc_base.get_params()\n#dtc_base_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:13.999797Z","iopub.execute_input":"2025-01-01T01:27:14.00062Z","iopub.status.idle":"2025-01-01T01:27:14.009448Z","shell.execute_reply.started":"2025-01-01T01:27:14.00058Z","shell.execute_reply":"2025-01-01T01:27:14.00858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#dtc0_params = dtc0.get_params()\n#dtc0_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.010615Z","iopub.execute_input":"2025-01-01T01:27:14.010929Z","iopub.status.idle":"2025-01-01T01:27:14.019721Z","shell.execute_reply.started":"2025-01-01T01:27:14.010902Z","shell.execute_reply":"2025-01-01T01:27:14.018841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\ndtc_param_grid = {\n                #'max_depth':[4,5,6,7,8,9], #or could say 'max_depth':range(1,10), but remember the deeper we go the more overfitting\n                #'criterion':['gini', 'entropy', 'log_loss'],\n                'min_samples_leaf': [ 20, 50]} #or again min_samples_leaf:range(1,5)\n                                                #min_samples_split:range(1,10)\n\n#Valid parameters are: ['ccp_alpha', 'class_weight', 'criterion', 'max_depth', 'max_features', 'max_leaf_nodes', 'min_impurity_decrease',\n# \n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.020837Z","iopub.execute_input":"2025-01-01T01:27:14.021185Z","iopub.status.idle":"2025-01-01T01:27:14.03218Z","shell.execute_reply.started":"2025-01-01T01:27:14.02116Z","shell.execute_reply":"2025-01-01T01:27:14.031083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n%%time\ndtc_model = dtc0\n#trial_params = deepcopy(dtc_base_params)\n#trial_model = DecisionTreeClassifier(**trial_params)\ntrial_model=dtc_base\ngrid_dict_dtc, grid_count = get_grid_cv(dtc0_params, dtc_param_grid, trial_model, dtc_model, 'dtc_grid')\ncombo_dict_dtc = base_dict_dtc | grid_dict_dtc\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.033479Z","iopub.execute_input":"2025-01-01T01:27:14.033864Z","iopub.status.idle":"2025-01-01T01:27:14.04588Z","shell.execute_reply.started":"2025-01-01T01:27:14.033827Z","shell.execute_reply":"2025-01-01T01:27:14.044965Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\ndef print_grid_acc(iters,the_dict, grid_name):\n    count = iters\n    \n    while count != 0:\n\n        print(f\"Coordinate descent {grid_name+str(count)} ROC score: {np.round(the_dict[grid_name+str(count)]['test roc auc score'],4)}\")\n        count = count -1\n        \n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.04711Z","iopub.execute_input":"2025-01-01T01:27:14.047448Z","iopub.status.idle":"2025-01-01T01:27:14.056303Z","shell.execute_reply.started":"2025-01-01T01:27:14.047422Z","shell.execute_reply":"2025-01-01T01:27:14.055311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#print_grid_acc(grid_count, grid_dict, 'dtc_grid_')\n#print_grid_score(grid_count, grid_dict_dtc, 'dtc_grid_', MY_SCORE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.057893Z","iopub.execute_input":"2025-01-01T01:27:14.058225Z","iopub.status.idle":"2025-01-01T01:27:14.064857Z","shell.execute_reply.started":"2025-01-01T01:27:14.058197Z","shell.execute_reply":"2025-01-01T01:27:14.063796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\ndefault_params_dtc = {}\n\nfor key in dtc0_params.keys():\n    default_params_dtc[key] = dtc0_params[key][0]\n\n#providing default parameters to xgbc model, before randomized search cross-validation\ndtc_rs_model = DecisionTreeClassifier(**default_params_dtc)\n\nrs_dict_dtc, dtc_rs_best_params, dtc_rs_model = get_random_searchcv(dtc_param_grid, dtc_rs_model, 'dtc_rs')\n\nresults_dict_dtc = combo_dict_dtc | rs_dict_dtc\n\nwith open(pickle_path+'dtc_rs.pickle', 'wb') as to_write:\n    pickle.dump(dtc_rs_model, to_write)\n'''    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.066162Z","iopub.execute_input":"2025-01-01T01:27:14.06649Z","iopub.status.idle":"2025-01-01T01:27:14.076334Z","shell.execute_reply.started":"2025-01-01T01:27:14.066464Z","shell.execute_reply":"2025-01-01T01:27:14.075464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#stop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.077563Z","iopub.execute_input":"2025-01-01T01:27:14.077943Z","iopub.status.idle":"2025-01-01T01:27:14.086423Z","shell.execute_reply.started":"2025-01-01T01:27:14.077908Z","shell.execute_reply":"2025-01-01T01:27:14.085612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\ndef print_grid_score(iters,the_dict, grid_name, which_score):\n    count = iters\n    \n    while count != 0:\n\n        print(f\"Coordinate descent {grid_name+str(count)} ROC score: {np.round(the_dict[grid_name+str(count)][which_score],4)}\")\n        count = count -1\n        \nMY_SCORE='test roc auc score'\n\nprint(f\"Benchmark ROC score: {np.round(results_dict_dtc['dtc0'][MY_SCORE],4)}\")\nprint_grid_score(grid_count, results_dict_dtc, 'dtc_grid_', MY_SCORE)\nprint(f\"Randomized search ROC score: {np.round(results_dict_dtc['dtc_rs_'][MY_SCORE],4)}\")\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.087615Z","iopub.execute_input":"2025-01-01T01:27:14.08789Z","iopub.status.idle":"2025-01-01T01:27:14.098714Z","shell.execute_reply.started":"2025-01-01T01:27:14.087867Z","shell.execute_reply":"2025-01-01T01:27:14.097839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n\n\n\nprint(f\"Benchmark ROC score: {np.round(results_dict_dtc['dtc0']['test roc auc score'],4)}\")\nprint_grid_acc(grid_count, results_dict_dtc, 'dtc_grid_')\nprint(f\"Randomized search ROC score: {np.round(results_dict_dtc['dtc_rs_']['test roc auc score'],4)}\")\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.099897Z","iopub.execute_input":"2025-01-01T01:27:14.100234Z","iopub.status.idle":"2025-01-01T01:27:14.112624Z","shell.execute_reply.started":"2025-01-01T01:27:14.100206Z","shell.execute_reply":"2025-01-01T01:27:14.111659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#dtc_rs_model.best_params_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.113867Z","iopub.execute_input":"2025-01-01T01:27:14.114262Z","iopub.status.idle":"2025-01-01T01:27:14.122051Z","shell.execute_reply.started":"2025-01-01T01:27:14.114225Z","shell.execute_reply":"2025-01-01T01:27:14.121083Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Random Forest**","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n%%time\n#what are bnasics for dtc?\n\nrfc_base = RandomForestClassifier()\nrfc_base.fit(X_train , y_train)\ntest_ras = roc_auc_score(y_test, rfc_base.predict_proba(X_test)[:,1])\n\nprint(f\"Base AUC score: {np.round(test_ras,4)}\")    \n\n\n\nrfc0 = RandomForestClassifier(n_estimators = 20, random_state = 42)\n\n                    \nrfc0.fit(X_train , y_train)\n\n\nrfc0_params = get_default_params(rfc0)\nbase_dict_rfc = {}\nbase_dict_rfc['rfc0'], rfc0_best_params = run_base_model(rfc0, rfc0_params)\n\nwith open(pickle_path+'rfc_clf0.pickle', 'wb') as to_write:\n    pickle.dump(rfc0, to_write)\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.123366Z","iopub.execute_input":"2025-01-01T01:27:14.124065Z","iopub.status.idle":"2025-01-01T01:27:14.135588Z","shell.execute_reply.started":"2025-01-01T01:27:14.124006Z","shell.execute_reply":"2025-01-01T01:27:14.134587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#print(f\"Benchmark AUC score: {np.round(base_dict_rfc['rfc0']['test roc auc score'],4)}\")  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.136792Z","iopub.execute_input":"2025-01-01T01:27:14.137112Z","iopub.status.idle":"2025-01-01T01:27:14.145339Z","shell.execute_reply.started":"2025-01-01T01:27:14.137086Z","shell.execute_reply":"2025-01-01T01:27:14.144432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#rfc_base_params = rfc_base.get_params()\n#rfc_base_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.146623Z","iopub.execute_input":"2025-01-01T01:27:14.146958Z","iopub.status.idle":"2025-01-01T01:27:14.154631Z","shell.execute_reply.started":"2025-01-01T01:27:14.146929Z","shell.execute_reply":"2025-01-01T01:27:14.153747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#rfc0_params = rfc0.get_params()\n#rfc0_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.155872Z","iopub.execute_input":"2025-01-01T01:27:14.156196Z","iopub.status.idle":"2025-01-01T01:27:14.165623Z","shell.execute_reply.started":"2025-01-01T01:27:14.156167Z","shell.execute_reply":"2025-01-01T01:27:14.164732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nrfc_param_grid ={\n                #'max_depth':[3,5,10,None],\n                #'n_estimators':[10,100,200],\n                #'max_features':[1,3,5,7],\n                'min_samples_leaf':[2,3]#,4,5,6],\n                #'min_samples_split':[2,3,4,5,6]\n           }\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.166975Z","iopub.execute_input":"2025-01-01T01:27:14.168008Z","iopub.status.idle":"2025-01-01T01:27:14.178606Z","shell.execute_reply.started":"2025-01-01T01:27:14.167976Z","shell.execute_reply":"2025-01-01T01:27:14.1776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n%%time\n\nrfc_model = rfc0\n\n#trial_params = deepcopy(rfc_base_params)\n#trial_model = RandomForestClassifier(**trial_params)\ntrial_model = rfc_base\n\ngrid_dict_rfc, grid_count = get_grid_cv(rfc0_params, rfc_param_grid, trial_model, rfc_model, 'rfc_grid')\n\ncombo_dict_rfc = base_dict_rfc | grid_dict_rfc\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.179888Z","iopub.execute_input":"2025-01-01T01:27:14.180804Z","iopub.status.idle":"2025-01-01T01:27:14.187749Z","shell.execute_reply.started":"2025-01-01T01:27:14.180766Z","shell.execute_reply":"2025-01-01T01:27:14.186846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\ndef print_grid_acc(iters,the_dict, grid_name):\n    count = iters\n    \n    while count != 0:\n\n        print(f\"Coordinate descent {grid_name+str(count)} ROC score: {np.round(the_dict[grid_name+str(count)]['test roc auc score'],4)}\")\n        count = count -1\n        \nprint_grid_acc(grid_count, grid_dict_rfc, 'rfc_grid_')\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.188868Z","iopub.execute_input":"2025-01-01T01:27:14.189174Z","iopub.status.idle":"2025-01-01T01:27:14.204558Z","shell.execute_reply.started":"2025-01-01T01:27:14.189147Z","shell.execute_reply":"2025-01-01T01:27:14.203502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#print_grid_score(grid_count, grid_dict_rfc, 'rfc_grid_', MY_SCORE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.20577Z","iopub.execute_input":"2025-01-01T01:27:14.206094Z","iopub.status.idle":"2025-01-01T01:27:14.213729Z","shell.execute_reply.started":"2025-01-01T01:27:14.206062Z","shell.execute_reply":"2025-01-01T01:27:14.212845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\ndefault_params_rfc = {}\n\nfor key in rfc0_params.keys():\n    default_params_rfc[key] = rfc0_params[key][0]\n\n#providing default parameters to xgbc model, before randomized search cross-validation\nrfc_rs_model = RandomForestClassifier(**default_params_rfc)\n\nrs_dict_rf, rfc_rs_best_params, rfc_rs_model = get_random_searchcv(rfc_param_grid, rfc_rs_model, 'rfc_rs')\n\nresults_dict_rfc = combo_dict_rfc | rs_dict_rfc\n\nwith open(pickle_path+'rfc_rs.pickle', 'wb') as to_write:\n    pickle.dump(rfc_rs_model, to_write)\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.214944Z","iopub.execute_input":"2025-01-01T01:27:14.215286Z","iopub.status.idle":"2025-01-01T01:27:14.225345Z","shell.execute_reply.started":"2025-01-01T01:27:14.215259Z","shell.execute_reply":"2025-01-01T01:27:14.224356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\ndef print_grid_score(iters,the_dict, grid_name, which_score):\n    count = iters\n    \n    while count != 0:\n\n        print(f\"Coordinate descent {grid_name+str(count)} ROC score: {np.round(the_dict[grid_name+str(count)][which_score],4)}\")\n        count = count -1\n '''       \n\n'''\nprint(f\"Benchmark ROC score: {np.round(results_dict_rfc['rfc0'][MY_SCORE],4)}\")\nprint_grid_score(grid_count, results_dict_rfc, 'rfc_grid_', MY_SCORE)\nprint(f\"Randomized search ROC score: {np.round(results_dict_rfc['rfc_rs_'][MY_SCORE],4)}\")\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.226532Z","iopub.execute_input":"2025-01-01T01:27:14.227253Z","iopub.status.idle":"2025-01-01T01:27:14.235642Z","shell.execute_reply.started":"2025-01-01T01:27:14.227198Z","shell.execute_reply":"2025-01-01T01:27:14.234676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\ndef print_grid_acc(iters,the_dict, grid_name):\n    count = iters\n    \n    while count != 0:\n\n        print(f\"Coordinate descent {grid_name+str(count)} ACC score: {np.round(the_dict[grid_name+str(count)]['test_accuracy'],4)}\")\n        count = count -1\n\nprint(f\"Benchmark ACC score: {np.round(results_dict_rfc['rfc0']['test_accuracy'],4)}\")\nprint_grid_acc(grid_count, results_dict_rfc, 'rfc_grid_')\nprint(f\"Randomized search ACC score: {np.round(results_dict_rfc['rfc_rs_']['test_accuracy'],4)}\")\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.23694Z","iopub.execute_input":"2025-01-01T01:27:14.237393Z","iopub.status.idle":"2025-01-01T01:27:14.245708Z","shell.execute_reply.started":"2025-01-01T01:27:14.237362Z","shell.execute_reply":"2025-01-01T01:27:14.244688Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **GradientBoostingClassifier**","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n%%time\n#what are bnasics for dtc?\n\ngbc_base = GradientBoostingClassifier()\ngbc_base.fit(X_train , y_train)\ntest_ras = roc_auc_score(y_test, gbc_base.predict_proba(X_test)[:,1])\n\nprint(f\"Base AUC score: {np.round(test_ras,4)}\")    \n\n\ngbc0 = GradientBoostingClassifier(n_estimators=50,random_state = 42)\ngbc0.fit(X_train , y_train)\n\ngbc0_params = get_default_params(gbc0)\nbase_dict_gbc = {}\nbase_dict_gbc['gbc0'], gbc0_best_params = run_base_model(gbc0, gbc0_params)\n\nwith open(pickle_path+'gbc_clf0.pickle', 'wb') as to_write:\n    pickle.dump(trial_model, to_write)\n \nprint(f\"Benchmark AUC score: {np.round(base_dict_gbc['gbc0']['test roc auc score'],4)}\") \n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.247141Z","iopub.execute_input":"2025-01-01T01:27:14.247833Z","iopub.status.idle":"2025-01-01T01:27:14.260378Z","shell.execute_reply.started":"2025-01-01T01:27:14.247796Z","shell.execute_reply":"2025-01-01T01:27:14.259354Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#gbc_base_params = gbc_base.get_params()\n#gbc_base_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.261831Z","iopub.execute_input":"2025-01-01T01:27:14.262731Z","iopub.status.idle":"2025-01-01T01:27:14.268278Z","shell.execute_reply.started":"2025-01-01T01:27:14.2627Z","shell.execute_reply":"2025-01-01T01:27:14.267411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\ngbc_param_grid = {\n    #\"loss\":[\"log_loss\"],\n    \"learning_rate\": [0.08,0.2],\n    #\"min_samples_split\": [2,3,4,5,6],\n    #\"min_samples_leaf\": [2,3,4,5,6],\n    #\"max_depth\":[9,10,11,12],\n    #\"max_features\":[\"sqrt\"],\n    #\"criterion\": [\"friedman_mse\", \"squared_error\" ],\n    #\"subsample\":[0.9],\n    #\"n_estimators\":[100,200]\n    }\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.269676Z","iopub.execute_input":"2025-01-01T01:27:14.270495Z","iopub.status.idle":"2025-01-01T01:27:14.280486Z","shell.execute_reply.started":"2025-01-01T01:27:14.270454Z","shell.execute_reply":"2025-01-01T01:27:14.279583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n%%time\n\ngbc_model = gbc0\n\ntrial_params = deepcopy(gbc_base_params)\ntrial_model = GradientBoostingClassifier(**trial_params)\ntrial_model=gbc_base\ngrid_dict_gbc, grid_count = get_grid_cv(gbc0_params, gbc_param_grid, trial_model, gbc_model, 'gbc_grid')\n\ncombo_dict_gbc = base_dict_gbc | grid_dict_gbc\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.281606Z","iopub.execute_input":"2025-01-01T01:27:14.281903Z","iopub.status.idle":"2025-01-01T01:27:14.29182Z","shell.execute_reply.started":"2025-01-01T01:27:14.281879Z","shell.execute_reply":"2025-01-01T01:27:14.291016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#print_grid_score(grid_count, grid_dict_gbc, 'gbc_grid_', MY_SCORE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.293162Z","iopub.execute_input":"2025-01-01T01:27:14.29378Z","iopub.status.idle":"2025-01-01T01:27:14.304192Z","shell.execute_reply.started":"2025-01-01T01:27:14.293743Z","shell.execute_reply":"2025-01-01T01:27:14.303194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\ndefault_params_gbc = {}\n\nfor key in gbc0_params.keys():\n    default_params_gbc[key] = gbc0_params[key][0]\n\n#providing default parameters to xgbc model, before randomized search cross-validation\ngbc_rs_model = GradientBoostingClassifier(**default_params_gbc)\n\nrs_dict, gbc_rs_best_params, gbc_rs_model = get_random_searchcv(gbc_param_grid, gbc_rs_model, 'gbc_rs')\n\nresults_dict_gbc = combo_dict_gbc | rs_dict_gbc\n\n\nwith open(pickle_path+'gbc_rs.pickle', 'wb') as to_write:\n    pickle.dump(gbc_rs_model, to_write)\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.305927Z","iopub.execute_input":"2025-01-01T01:27:14.306345Z","iopub.status.idle":"2025-01-01T01:27:14.317334Z","shell.execute_reply.started":"2025-01-01T01:27:14.306308Z","shell.execute_reply":"2025-01-01T01:27:14.31633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n\nprint(f\"Benchmark ROC score: {np.round(results_dict_gbc['gbc0'][MY_SCORE],4)}\")\nprint_grid_score(grid_count, results_dict_gbc, 'gbc_grid_', MY_SCORE)\nprint(f\"Randomized search ROC score: {np.round(results_dict_gbc['gbc_rs_'][MY_SCORE],4)}\")\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.318756Z","iopub.execute_input":"2025-01-01T01:27:14.31908Z","iopub.status.idle":"2025-01-01T01:27:14.334892Z","shell.execute_reply.started":"2025-01-01T01:27:14.319043Z","shell.execute_reply":"2025-01-01T01:27:14.334021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **HistGradientBoostingClassifier**","metadata":{}},{"cell_type":"code","source":"#USE INSTEAD OF GradientBoostingClassifier FOR LARGER DATA SETS","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.336174Z","iopub.execute_input":"2025-01-01T01:27:14.336557Z","iopub.status.idle":"2025-01-01T01:27:14.34208Z","shell.execute_reply.started":"2025-01-01T01:27:14.336521Z","shell.execute_reply":"2025-01-01T01:27:14.341193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n%%time\n#what are bnasics for dtc?\n#missed this in docs, but cat features :\"expected to have a cardinality <= 255\"\n#so I could bin a few of these in this particular case and then make cats or just treat them as floats and transform\nhgbc_base = HistGradientBoostingClassifier(\n                categorical_features=categorical_features_indices,\n                random_state=42) \nhgbc_base.fit(X_train , y_train)\ntest_ras = roc_auc_score(y_test, hgbc_base.predict_proba(X_test)[:,1])\n\nprint(f\"Base AUC score: {np.round(test_ras,4)}\")    \n\nhgbc0 = HistGradientBoostingClassifier(\n                loss='log_loss',\n                categorical_features=categorical_features_indices,\n                early_stopping=True,\n                n_iter_no_change=10, #no of rounds if early stopping is enabled\n                scoring='loss',\n                max_iter=500, #default = 100\n#Proportion (or absolute size) of training data to set aside as validation data for early stopping. \n#If None, early stopping is done on the training data. Only used if early stopping is performed.\n                validation_fraction=0.2, \n                random_state=42) \nhgbc0.fit(X_train , y_train)\n\nhgbc0_params = get_default_params(hgbc0)\nbase_dict_hgbc = {}\nbase_dict_hgbc['hgbc0'], hgbc0_best_params = run_base_model(hgbc0, hgbc0_params)\n\nwith open(pickle_path+'hgbc_clf0.pickle', 'wb') as to_write:\n    pickle.dump(hgbc0, to_write)\n\nprint(f\"Benchmark AUC score: {np.round(base_dict_hgbc['hgbc0']['test roc auc score'],4)}\")\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.343219Z","iopub.execute_input":"2025-01-01T01:27:14.34357Z","iopub.status.idle":"2025-01-01T01:27:14.354315Z","shell.execute_reply.started":"2025-01-01T01:27:14.343544Z","shell.execute_reply":"2025-01-01T01:27:14.353383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#hgbc_base_params = hgbc_base.get_params()\n#hgbc_base_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.355431Z","iopub.execute_input":"2025-01-01T01:27:14.355764Z","iopub.status.idle":"2025-01-01T01:27:14.365351Z","shell.execute_reply.started":"2025-01-01T01:27:14.355737Z","shell.execute_reply":"2025-01-01T01:27:14.364433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nclass sklearn.ensemble.HistGradientBoostingClassifier(loss='auto', *, learning_rate=0.1, max_iter=100,\nmax_leaf_nodes=31, max_depth=None,min_samples_leaf=20, l2_regularization=0.0,\nmax_bins=255, categorical_features=None, monotonic_cst=None, warm_start=False, early_stopping='auto',\nscoring='loss', validation_fraction=0.1, n_iter_no_change=10, tol=1e-07, verbose=0, random_state=None)\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.366899Z","iopub.execute_input":"2025-01-01T01:27:14.367327Z","iopub.status.idle":"2025-01-01T01:27:14.3776Z","shell.execute_reply.started":"2025-01-01T01:27:14.36729Z","shell.execute_reply":"2025-01-01T01:27:14.376767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nhgbc_param_grid = {\n        #'max_iter': [100,500,1000,2000],\n        #'learning_rate': [0.08,0.2],\n        #'max_depth': [25, 50, 75, 150],\n        'min_samples_leaf':[10,30,], #DEFAULT=20\n        #'max_leaf_nodes':[] #int or None, default = 31, If None, there is no maximum limit.\n        #'l2_regularization': [0, 0.1, 0.5, 1, 1.5, 6],\n        #\n        #'max_features':[0.8,0.9] #default 1 BUT DOESNT SEEMED TO BE ALLOWED! THROWS ERROR NOT ONE OF THE PARAMETERS! WHY DOCS SHOW IT EXISTS MAYBE UPDATE SKLEARN??\n        }\n#max_features sounds like col_by_tree from xgboost\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.378779Z","iopub.execute_input":"2025-01-01T01:27:14.379074Z","iopub.status.idle":"2025-01-01T01:27:14.391354Z","shell.execute_reply.started":"2025-01-01T01:27:14.379042Z","shell.execute_reply":"2025-01-01T01:27:14.390466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n%%time\n\nhgbc_model = hgbc0\n\n#trial_params = deepcopy(hgbc_base_params)\n#trial_model = HistGradientBoostingClassifier(**trial_params)\ntrial_model = hgbc_base\ngrid_dict_hgbc, grid_count = get_grid_cv(hgbc0_params, hgbc_param_grid, trial_model, hgbc_model, 'hgbc_grid')\n\ncombo_dict_hgbc = base_dict_hgbc | grid_dict_hgbc\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.392554Z","iopub.execute_input":"2025-01-01T01:27:14.392866Z","iopub.status.idle":"2025-01-01T01:27:14.408549Z","shell.execute_reply.started":"2025-01-01T01:27:14.39284Z","shell.execute_reply":"2025-01-01T01:27:14.407623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#print_grid_score(grid_count, grid_dict_hgbc, 'hgbc_grid_', MY_SCORE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.409744Z","iopub.execute_input":"2025-01-01T01:27:14.410071Z","iopub.status.idle":"2025-01-01T01:27:14.416314Z","shell.execute_reply.started":"2025-01-01T01:27:14.410017Z","shell.execute_reply":"2025-01-01T01:27:14.4153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#grid_model = pickle.load(open('/kaggle/working/hgbc_grid_1.pickle', 'rb'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.417744Z","iopub.execute_input":"2025-01-01T01:27:14.418082Z","iopub.status.idle":"2025-01-01T01:27:14.426868Z","shell.execute_reply.started":"2025-01-01T01:27:14.418044Z","shell.execute_reply":"2025-01-01T01:27:14.425893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#grid_model.best_params_\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.428069Z","iopub.execute_input":"2025-01-01T01:27:14.428386Z","iopub.status.idle":"2025-01-01T01:27:14.435852Z","shell.execute_reply.started":"2025-01-01T01:27:14.42836Z","shell.execute_reply":"2025-01-01T01:27:14.43506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#grid_model.best_score_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.437003Z","iopub.execute_input":"2025-01-01T01:27:14.437361Z","iopub.status.idle":"2025-01-01T01:27:14.448434Z","shell.execute_reply.started":"2025-01-01T01:27:14.437333Z","shell.execute_reply":"2025-01-01T01:27:14.447465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#grid_model.cv_results_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.449805Z","iopub.execute_input":"2025-01-01T01:27:14.450249Z","iopub.status.idle":"2025-01-01T01:27:14.459803Z","shell.execute_reply.started":"2025-01-01T01:27:14.450208Z","shell.execute_reply":"2025-01-01T01:27:14.458884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#grid_model.cv_results_['params']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.460964Z","iopub.execute_input":"2025-01-01T01:27:14.461303Z","iopub.status.idle":"2025-01-01T01:27:14.471568Z","shell.execute_reply.started":"2025-01-01T01:27:14.461275Z","shell.execute_reply":"2025-01-01T01:27:14.470699Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#grid_model.cv_results_['mean_test_score']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.472856Z","iopub.execute_input":"2025-01-01T01:27:14.473528Z","iopub.status.idle":"2025-01-01T01:27:14.484518Z","shell.execute_reply.started":"2025-01-01T01:27:14.473489Z","shell.execute_reply":"2025-01-01T01:27:14.483588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\ndefault_params_hgbc = {}\n\nfor key in hgbc0_params.keys():\n    default_params_hgbc[key] = hgbc0_params[key][0]\n\n#providing default parameters to xgbc model, before randomized search cross-validation\nhgbc_rs_model = HistGradientBoostingClassifier(**default_params_hgbc)\n\nrs_dict_hgbc, hgbc_rs_best_params, hgbc_rs_model = get_random_searchcv(hgbc_param_grid, hgbc_rs_model, 'hgbc_rs')\n\nresults_dict_hgbc = combo_dict_hgbc | rs_dict_hgbc\n\n\nwith open(pickle_path+'hgbc_rs.pickle', 'wb') as to_write:\n    pickle.dump(hgbc_rs_model, to_write)\n'''   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.485617Z","iopub.execute_input":"2025-01-01T01:27:14.485933Z","iopub.status.idle":"2025-01-01T01:27:14.496946Z","shell.execute_reply.started":"2025-01-01T01:27:14.485907Z","shell.execute_reply":"2025-01-01T01:27:14.496199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n\nprint(f\"Benchmark ROC score: {np.round(results_dict_hgbc['hgbc0'][MY_SCORE],4)}\")\nprint_grid_score(grid_count, results_dict_hgbc, 'hgbc_grid_', MY_SCORE)\nprint(f\"Randomized search ROC score: {np.round(results_dict_hgbc['hgbc_rs_'][MY_SCORE],4)}\")\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.498049Z","iopub.execute_input":"2025-01-01T01:27:14.498368Z","iopub.status.idle":"2025-01-01T01:27:14.50798Z","shell.execute_reply.started":"2025-01-01T01:27:14.498343Z","shell.execute_reply":"2025-01-01T01:27:14.507083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#results_dict_hgbc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.509121Z","iopub.execute_input":"2025-01-01T01:27:14.50941Z","iopub.status.idle":"2025-01-01T01:27:14.520877Z","shell.execute_reply.started":"2025-01-01T01:27:14.509384Z","shell.execute_reply":"2025-01-01T01:27:14.520076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **XGBOOST**","metadata":{}},{"cell_type":"code","source":"stop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.521988Z","iopub.execute_input":"2025-01-01T01:27:14.522341Z","iopub.status.idle":"2025-01-01T01:27:14.878593Z","shell.execute_reply.started":"2025-01-01T01:27:14.522315Z","shell.execute_reply":"2025-01-01T01:27:14.870327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n%%time\n##########################################\n#what Im trying to do, no extra params\nxgbc_base = XGBRegressor(\n                        objective='reg:squaredlogerror',\n                        enable_categorical=True,\n                        max_bin=60,#Age is biggest at 47\n                        n_estimators=100, #default is 100\n                        eval_metric='rmsle', #\n                        device='cuda',\n                        random_state=42,\n                         )\neval_set = [(X_test, y_test)]\nxgbc_base.fit(X_train , y_train, verbose=0) #eval_set=eval_set, early_stopping_rounds=10,\ntest_predictions = xgbc_base.predict(X_test)\n\n#orig_sz_pred = np.expm1(test_predictions) #do at end?\ntest_rmsle = rmsle_score(y_test, test_predictions)#[:,1])\n\n\nprint(f\"Base RMSLE score: {np.round(test_rmsle,6)}\") \n\n#Base RMSLE score: 1.062788\n#CPU times: user 4.89 s, sys: 94.9 ms, total: 4.99 s\n#Wall time: 1.92 s","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.87948Z","iopub.status.idle":"2025-01-01T01:27:14.879853Z","shell.execute_reply.started":"2025-01-01T01:27:14.879685Z","shell.execute_reply":"2025-01-01T01:27:14.879702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.881144Z","iopub.status.idle":"2025-01-01T01:27:14.881508Z","shell.execute_reply.started":"2025-01-01T01:27:14.881339Z","shell.execute_reply":"2025-01-01T01:27:14.881357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#print(xgb.__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.883083Z","iopub.status.idle":"2025-01-01T01:27:14.883569Z","shell.execute_reply.started":"2025-01-01T01:27:14.883318Z","shell.execute_reply":"2025-01-01T01:27:14.883342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n%%time\n############################################\n#in this I want to add a few params I think might be basics and that I will then do a grid on\n#basically putting back in basics as they come out of eith refining grid or random search\n#start by changing the default lr from 0.3 default\n#and start with high estimators and set early stopping rounds in fit to sat 100\nxgbc0 = XGBRegressor(\n                        objective='reg:squarederror', \n                        enable_categorical=True,\n                        max_bin=50,#\n                        eval_metric='rmsle', \n                        device='cuda',\n                        random_state=42,\n                        #booster='gbtree',\n\n                        \n                        #max_cat_to_onehot='15', #varied btw 4 and 15\n\n                        max_depth=5,#def 6\n\n                        min_child_weight=2,#def 1\n                        gamma=1e-9, #0.4,#5e-6, 0.7\n                        colsample_bytree=1, #0.7,\n                        subsample=1, #0.9, \n    \n                        learning_rate=0.05,#def 0.3\n                        n_estimators=1000, #default is 100\n\n                        reg_alpha=10, #l1 def=0\n                        reg_lambda=0.01, #l2 def=1 10 worked well too\n                        \n                    ###################################################\n\n                        #tree_method='hist',\n                        #grow_policy='depthwise',\n\n                        )\n\n\n\n#xgbc0.fit(X_train , y_train, verbose=0)# eval_set=eval_set, early_stopping_rounds=100,verbose=0)\n#note I dont really need this step, since its doing a  cv search then later grid search ,\n#xgbc0_params = get_default_params(xgbc0)\nbase_dict_xgbc = {}\n#print(f\"Type of base_dict_xgbc['xgbc0'] before the function call: {type(base_dict_xgbc.get('xgbc0', 'Not Found'))}\")\n#print(f\"Model passed into function is of type: {type(xgbc0)}\") # Check type here\n#suggestion from genini\n#base_dict_xgbc['xgbc0'] = xgbc0 #Store the xgbc0 instance\n#im instead just renaming from 'xgbc0' to 'xgbc'\nbase_dict_xgbc['xgbc0'], xgbc_best_params = run_base_model(xgbc0)#, xgbc0_params)\n\n\nwith open(pickle_path+'xgbc_clf0.pickle', 'wb') as to_write:\n    pickle.dump(xgbc0, to_write)\n\n\nprint(f\"Benchmark RMSLE score: {np.round(base_dict_xgbc['xgbc0']['rmsle'],6)}\")\nprint(f\"Benchmark CV_MEAN_RMSLE score: {np.round(base_dict_xgbc['xgbc0']['std_cv'],6)}\")\nprint(f\"Benchmark CV_MEAN score: {np.round(base_dict_xgbc['xgbc0']['mean_cv'],6)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:28:04.18468Z","iopub.execute_input":"2025-01-01T01:28:04.185085Z","iopub.status.idle":"2025-01-01T01:33:19.982525Z","shell.execute_reply.started":"2025-01-01T01:28:04.185045Z","shell.execute_reply":"2025-01-01T01:33:19.981523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nCross-validation scores: [-1.05934715 -1.05159213 -1.04940477 -1.05658177 -1.05859436 -1.06083881\n -1.05980932 -1.05238827]\nMean CV score: -1.05606957098445\nThe test RMSLE score: 1.058125411329247\nBenchmark RMSLE score: 1.058125\nBenchmark CV_MEAN_RMSLE score: 0.004064\nBenchmark CV_MEAN score: -1.05607\nCPU times: user 1min 28s, sys: 2.29 s, total: 1min 31s\nWall time: 1min 5s\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.887307Z","iopub.status.idle":"2025-01-01T01:27:14.887798Z","shell.execute_reply.started":"2025-01-01T01:27:14.887537Z","shell.execute_reply":"2025-01-01T01:27:14.88756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(base_dict_xgbc['xgbc0']['cv_results'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.888891Z","iopub.status.idle":"2025-01-01T01:27:14.889244Z","shell.execute_reply.started":"2025-01-01T01:27:14.889078Z","shell.execute_reply":"2025-01-01T01:27:14.889096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nrandom_states = [42, 123, 456, 789, 1213]\nthe_res_x, the_best_rmsle_x, the_rand_st_x= run_base_model_many_splits(xgbc0, X, Y, 0.25, random_states)\n\nprint(the_best_rmsle_x)\nprint(the_rand_st_x)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.890668Z","iopub.status.idle":"2025-01-01T01:27:14.891003Z","shell.execute_reply.started":"2025-01-01T01:27:14.89084Z","shell.execute_reply":"2025-01-01T01:27:14.890857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_importance_df= xgbc0.feature_importances_\n#feature_importance_df= xgbc0.get_feature_importance(prettified=True)\nfeature_importance_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.89193Z","iopub.status.idle":"2025-01-01T01:27:14.892271Z","shell.execute_reply.started":"2025-01-01T01:27:14.892113Z","shell.execute_reply":"2025-01-01T01:27:14.892129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nplt.figure(figsize=(12, 6));\nsns.barplot(x=\"Importances\", y=\"Feature Id\", data=feature_importance_df);\nplt.title('CatBoost features importance:');\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.893202Z","iopub.status.idle":"2025-01-01T01:27:14.893515Z","shell.execute_reply.started":"2025-01-01T01:27:14.893355Z","shell.execute_reply":"2025-01-01T01:27:14.89337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n# plot\nplt.bar(range(len(feature_importance_df)), feature_importance_df)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.894565Z","iopub.status.idle":"2025-01-01T01:27:14.894893Z","shell.execute_reply.started":"2025-01-01T01:27:14.894735Z","shell.execute_reply":"2025-01-01T01:27:14.894752Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nxgbc_param_grid = {\n                #'n_estimators': [1000, 1200,1500],#def=100\n                #'learning_rate': [0.1,0.05,0.01], #def=0.3\n    \n                #'max_depth':[5,6,7,8], #def=6\n                #'max_depth':range(5,24,2),\n                #'min_child_weight':range(1,3,1),\n                #'min_child_weight':[1,2,3],#def=1 (higher less overfit)0-> inf\n                #'subsample':[0.9, 1],#def=1\n                #'colsample_bytree':[0.7, 0.8,0.9,1],#def=1\n                #'subsample':[i/10.0 for i in range(5,10)],\n                #'colsample_bytree':[i/10.0 for i in range(3,10)]\n\n                #'gamma':[0.6, 0.7, 0.8], #def=0 0 -> inf (higher less overfit)\n                #'gamma':[i/10.0 for i in range(1,9)]\n                #'gamma': [1e-9, 1e-8, 1e-7, 1e-6,] #these gave same results so 5e-6 used midpoint\n                #'reg_lambda':[2,3, 10],#def=1 (higher less overfit)\n\n                #'learning_rate':[i/10.0 for i in range(1,4)],\n                #'learning_rat':[i/100.0 for i in range(75,90,5)],\n                #'learning_rate':[0.1,0.2, 0.25],\n                #'scale_pos_weight':[1,2.670254945755005,7],\n                \n                'reg_lambda': [1e-6, 1e-2, 1, 10], #1\n                #'reg_lambda': [1, 10], \n                'reg_alpha': [1e-6, 1e-2, 1, 10] #0\n                #'reg_alpha': [1e-10, 1e-8],\n                }\n\n\n'''\n#start wide then narrow down\nparam_test1 = {\n 'max_depth':range(5,10,2),\n 'min_child_weight':range(1,6,2)\n}\nparam_test2 = {\n 'max_depth':[4,5,6],\n 'min_child_weight':[4,5,6]\n}\nparam_test2b = {\n 'min_child_weight':[6,8,10,12]\n}\n#if we get param values doing the best at the extreme of range try going pat that value - eg above say best min child weight is 6 so try again but higher range\n# now try gamma\nparam_test3 = {\n 'gamma':[i/10.0 for i in range(0,5)]\n}\n#once we have these three try above agian\n\nparam_test4 = {\n 'subsample':[i/10.0 for i in range(5,10)],\n 'colsample_bytree':[i/10.0 for i in range(3,10)]\n}\n#now narrow them\nparam_test5 = {\n 'subsample':[i/100.0 for i in range(75,90,5)],\n 'colsample_bytree':[i/100.0 for i in range(75,90,5)]\n}\n#regularizaTion params\nparam_test6 = {\n 'reg_alpha':[1e-5, 1e-2, 0.1, 1, 100]\n}\n#what value was the best?\n#if its at the extreme or where was the best score try arounde that, lets say best result happened when alpha = 0.01, so try\nparam_test7 = {\n 'reg_alpha':[0, 0.001, 0.005, 0.01, 0.05]\n}\n#input new values again above\n#now go back to lr try lowering or set a range in grid\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.896741Z","iopub.status.idle":"2025-01-01T01:27:14.897242Z","shell.execute_reply.started":"2025-01-01T01:27:14.896969Z","shell.execute_reply":"2025-01-01T01:27:14.896992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n%%time\n\n\n#xgbc_params = deepcopy(xgbc0_best_params)\nxgbc_model = xgbc0\n############################################\n#rial_params= deepcopy(xgbc_base_params)\n#trial_model = XGBClassifier(**trial_params)\n#trial_model = xgbc_base\n###########################################\n#grid_dict_xgbc , grid_count = get_grid_cv(xgbc0_params, xgbc_param_grid, trial_model, xgbc_model, 'xgbc_grid')\ngrid_dict_xgbc , grid_model = get_grid_cv(xgbc_param_grid, xgbc_model, 'xgbc_grid')\n#combo_dict_xgbc = base_dict_xgbc | grid_dict_xgbc\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.89864Z","iopub.status.idle":"2025-01-01T01:27:14.899142Z","shell.execute_reply.started":"2025-01-01T01:27:14.898875Z","shell.execute_reply":"2025-01-01T01:27:14.898899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#print_grid_acc(grid_count, grid_dict, 'xgbc_grid_')\n#print_grid_score(grid_count, grid_dict_xgbc, 'xgbc_grid_', MY_SCORE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.900985Z","iopub.status.idle":"2025-01-01T01:27:14.901479Z","shell.execute_reply.started":"2025-01-01T01:27:14.90123Z","shell.execute_reply":"2025-01-01T01:27:14.901254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#grid_model = pickle.load(open('/kaggle/working/xgbc_grid_1.pickle', 'rb'))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.902763Z","iopub.status.idle":"2025-01-01T01:27:14.903258Z","shell.execute_reply.started":"2025-01-01T01:27:14.90299Z","shell.execute_reply":"2025-01-01T01:27:14.903013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Benchmark RMSLE score: {np.round(grid_dict_xgbc['xgbc_grid_1']['rmsle'],6)}\")\n\n#print(f\"Benchmark CV_MEAN score: {np.round(grid_dict_xgbc['xgbc_grid_1']['mean_cv'],6)}\") This is pointless its just a mean of the means!","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.904866Z","iopub.status.idle":"2025-01-01T01:27:14.90527Z","shell.execute_reply.started":"2025-01-01T01:27:14.905091Z","shell.execute_reply":"2025-01-01T01:27:14.905111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grid_model.best_params_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.906326Z","iopub.status.idle":"2025-01-01T01:27:14.906683Z","shell.execute_reply.started":"2025-01-01T01:27:14.906519Z","shell.execute_reply":"2025-01-01T01:27:14.906535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grid_model.best_score_\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.908145Z","iopub.status.idle":"2025-01-01T01:27:14.908488Z","shell.execute_reply.started":"2025-01-01T01:27:14.908318Z","shell.execute_reply":"2025-01-01T01:27:14.908335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grid_model.cv_results_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.909936Z","iopub.status.idle":"2025-01-01T01:27:14.910444Z","shell.execute_reply.started":"2025-01-01T01:27:14.910185Z","shell.execute_reply":"2025-01-01T01:27:14.91021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#The big grid - no resourses without gpu see start above\n#With kaggle resourses Ive found these large grids just hard to do\n#its more efficient for me to incrmentally go a couple of parameters at a time and wath the cv sores and then keep incrementing\n'''\n\nxgbc_param_grid = {\n                #'gamma': [0,0.0001, 0.001, 0.1,0.2,0.8,3.2,12.8,51.2,102.4, 200], #larger means more conservative\n              #'learning_rate': [0.0001, 0.001,0.005, 0.01, 0.05, 0.1, 0.2, 0.5],\n              #'learning_rate': [0.0001, 0.001,0.005, 0.01, 0.05, 0.1,0.15, 0.2, 0.25, 0.5],\n              #'max_depth': [4,5,6,7,8,9], #larger = prone overfit, def =6\n              'max_depth': [3,4,5,6,7,],\n              #'n_estimators': [50,100,200, 1000],\n              #'n_estimators': [50,100,500,],\n              #'reg_alpha': [0,0.1,0.2,0.8,3.2,12.8,51.2,102.4,200], #higher more conservative\n              #'reg_lambda': [0,0.1,0.2,0.8,3.2,12.8,51.2,102.4,200], #higher more conservative\n              #'reg_alpha': [0,0.1,0.2,0.8,3,12,51,102], #higher more conservative\n              #'reg_lambda': [0,0.1,0.2,0.8,3,12,51,102], #higher more conservative\n              #'min_child_weight': [3,4,5,6], #def 1, larger more conservative\n              'min_child_weight': [5,10,],\n              'subsample':[0.7,0.8,0.9,1],  #amount of data sampled before growing trees\n              'colsample_bytree':[0.1,0.2,0.5,],\n              #'scale_pos_weight':[3,3.7259, 4, 5],\n              #'objective': ['binary:logistic'],\n              #'eval_metric':['auc'], \n              'grow_policy':['depthwise', 'lossguide'],\n              #'booster':['gbtree'], \n              #'tree_method':['hist'],\n              #'enable_categorical':True,\n              #'random_state':[42]\n                }\n              \n\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.911791Z","iopub.status.idle":"2025-01-01T01:27:14.912161Z","shell.execute_reply.started":"2025-01-01T01:27:14.911964Z","shell.execute_reply":"2025-01-01T01:27:14.911981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#combo_dict\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.913498Z","iopub.status.idle":"2025-01-01T01:27:14.913971Z","shell.execute_reply.started":"2025-01-01T01:27:14.913727Z","shell.execute_reply":"2025-01-01T01:27:14.913751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#maybe I could send in larger param grids to this and divide my iterations???\n'''\ndefault_params_xgbc = {}\n\n#for key in xgbc_params.keys():\nfor key in xgbc0_params.keys():\n    default_params_xgbc[key] = xgbc0_params[key][0]\n'''\n#providing default parameters to xgbc model, before randomized search cross-validation\n#xgbc_rs_model = XGBClassifier(**default_params_xgbc)\n#or just use the xgbc0 model from above instead\nxgbc_rs_model = xgbc0\nrs_dict_xgbc, xgbc_rs_best_params, xgbc_rs_model = get_random_searchcv(xgbc_param_grid, xgbc_rs_model, 'xgbc_rs')\n\n#results_dict_xgbc = combo_dict_xgbc | rs_dict_xgbc\n\nwith open(pickle_path+'xgbc_rs.pickle', 'wb') as to_write:\n    pickle.dump(xgbc_rs_model, to_write)\nprint(f\"Benchmark RMSLE score: {np.round(rs_dict_xgbc['xgbc_rs_']['rmsle'],6)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.915601Z","iopub.status.idle":"2025-01-01T01:27:14.916098Z","shell.execute_reply.started":"2025-01-01T01:27:14.915829Z","shell.execute_reply":"2025-01-01T01:27:14.915852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgbc_rs_model.best_params_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.917658Z","iopub.status.idle":"2025-01-01T01:27:14.918154Z","shell.execute_reply.started":"2025-01-01T01:27:14.917883Z","shell.execute_reply":"2025-01-01T01:27:14.917907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgbc_rs_model.best_score_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.919776Z","iopub.status.idle":"2025-01-01T01:27:14.920273Z","shell.execute_reply.started":"2025-01-01T01:27:14.920004Z","shell.execute_reply":"2025-01-01T01:27:14.920046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgbc_rs_model.cv_results_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.922083Z","iopub.status.idle":"2025-01-01T01:27:14.922433Z","shell.execute_reply.started":"2025-01-01T01:27:14.922264Z","shell.execute_reply":"2025-01-01T01:27:14.922281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nprint(f\"Benchmark ROC score: {np.round(results_dict_xgbc['xgbc0'][MY_SCORE],4)}\")\nprint_grid_score(grid_count, results_dict_xgbc, 'xgbc_grid_', MY_SCORE)\nprint(f\"Randomized search ROC score: {np.round(results_dict_xgbc['xgbc_rs_'][MY_SCORE],4)}\")\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.924425Z","iopub.status.idle":"2025-01-01T01:27:14.924735Z","shell.execute_reply.started":"2025-01-01T01:27:14.924583Z","shell.execute_reply":"2025-01-01T01:27:14.924598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#rand_model = pickle.load(open('/kaggle/working/xgbc_rs.pickle', 'rb'))\n#rand_model.best_params_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.925837Z","iopub.status.idle":"2025-01-01T01:27:14.926185Z","shell.execute_reply.started":"2025-01-01T01:27:14.925993Z","shell.execute_reply":"2025-01-01T01:27:14.926012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#results_dict_xgbc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.927326Z","iopub.status.idle":"2025-01-01T01:27:14.927642Z","shell.execute_reply.started":"2025-01-01T01:27:14.927484Z","shell.execute_reply":"2025-01-01T01:27:14.9275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# feature importance\n\n#print(xgbc_rs_model.best_estimator_.feature_importances_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.928976Z","iopub.status.idle":"2025-01-01T01:27:14.929327Z","shell.execute_reply.started":"2025-01-01T01:27:14.929171Z","shell.execute_reply":"2025-01-01T01:27:14.929187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nplt.bar(range(len(xgbc_rs_model.best_estimator_.feature_importances_)), xgbc_rs_model.best_estimator_.feature_importances_)\nplt.show()\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.930667Z","iopub.status.idle":"2025-01-01T01:27:14.930974Z","shell.execute_reply.started":"2025-01-01T01:27:14.930825Z","shell.execute_reply":"2025-01-01T01:27:14.93084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nfrom xgboost import plot_importance\nplot_importance(xgbc_rs_model.best_estimator_)\nplt.show()\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.93191Z","iopub.status.idle":"2025-01-01T01:27:14.932262Z","shell.execute_reply.started":"2025-01-01T01:27:14.932091Z","shell.execute_reply":"2025-01-01T01:27:14.932108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#stop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.933114Z","iopub.status.idle":"2025-01-01T01:27:14.933425Z","shell.execute_reply.started":"2025-01-01T01:27:14.933266Z","shell.execute_reply":"2025-01-01T01:27:14.933281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#SHAP values \n#import shap\n'''\nexplainer = shap.Explainer(xgbc_rs_model)\nshap_values = explainer(X_test)\nshap_importance = shap_values.abs.mean(0).values\nsorted_idx = shap_importance.argsort()\nfig = plt.figure(figsize=(12, 6))\nplt.barh(range(len(sorted_idx)), shap_importance[sorted_idx], align='center')\nplt.yticks(range(len(sorted_idx)), np.array(X_test.columns)[sorted_idx])\nplt.title('SHAP Importance')\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.935278Z","iopub.status.idle":"2025-01-01T01:27:14.935763Z","shell.execute_reply.started":"2025-01-01T01:27:14.935518Z","shell.execute_reply":"2025-01-01T01:27:14.935543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#alternative to SHAP\n#import sklearn.inspection.permutation_importance\n'''\nperm_importance = permutation_importance(xgbc_rs_model, X_test, y_test, n_repeats=10, random_state=42)\nsorted_idx = perm_importance.importances_mean.argsort()\nfig = plt.figure(figsize=(12, 6))\nplt.barh(range(len(sorted_idx)), perm_importance.importances_mean[sorted_idx], align='center')\nplt.yticks(range(len(sorted_idx)), np.array(X_test.columns)[sorted_idx])\nplt.title('Permutation Importance')\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.937238Z","iopub.status.idle":"2025-01-01T01:27:14.937717Z","shell.execute_reply.started":"2025-01-01T01:27:14.937472Z","shell.execute_reply":"2025-01-01T01:27:14.937497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#results_dict_xgbc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.938665Z","iopub.status.idle":"2025-01-01T01:27:14.939074Z","shell.execute_reply.started":"2025-01-01T01:27:14.938844Z","shell.execute_reply":"2025-01-01T01:27:14.93886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#results_dict = results_dict_xgbc | results_dict_lgbc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.941062Z","iopub.status.idle":"2025-01-01T01:27:14.941536Z","shell.execute_reply.started":"2025-01-01T01:27:14.941283Z","shell.execute_reply":"2025-01-01T01:27:14.941306Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#results_dict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.942558Z","iopub.status.idle":"2025-01-01T01:27:14.943017Z","shell.execute_reply.started":"2025-01-01T01:27:14.942778Z","shell.execute_reply":"2025-01-01T01:27:14.942803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"####################################END  XGBOOST CLASSIFIER ###############################################","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.944463Z","iopub.status.idle":"2025-01-01T01:27:14.944927Z","shell.execute_reply.started":"2025-01-01T01:27:14.944687Z","shell.execute_reply":"2025-01-01T01:27:14.944711Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"#stop","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.946323Z","iopub.status.idle":"2025-01-01T01:27:14.946695Z","shell.execute_reply.started":"2025-01-01T01:27:14.946513Z","shell.execute_reply":"2025-01-01T01:27:14.946531Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Cat Boost**","metadata":{}},{"cell_type":"code","source":"######################################START  ###############################################","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.948444Z","iopub.status.idle":"2025-01-01T01:27:14.948978Z","shell.execute_reply.started":"2025-01-01T01:27:14.948625Z","shell.execute_reply":"2025-01-01T01:27:14.948649Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n%%time\nfrom catboost import CatBoostClassifier\n\n\ncbc_3 = CatBoostClassifier(\n        loss_function='Logloss',\n        auto_class_weights='SqrtBalanced', \n        eval_metric='AUC',\n        iterations=100,\n        verbose=False,\n        #task_type='GPU',\n        #border_count= 32, # a recommendation to speed up the process\n        #early_stopping_rounds=100,                 \n        random_seed=42,)\n\n'''\n'''\ncbc0_1.fit(X_train, y_train,\n          eval_set=(X_valid, y_valid),\n          use_best_model=True,\n          plot=True\n         );\n\n'''\n'''\n#using pool\nfrom catboost import Pool\n\ntrain_data = Pool(data=X_train,\n                  label=y_train,\n                  cat_features=categorical_features_indices\n                 )\n\nvalid_data = Pool(data=X_test,\n                  label=y_test,\n                  cat_features=categorical_features_indices\n                 )\n\n#cbc_0_3=SAME AS ABOVE SAY for cbc0_1 - ie define catboost witrh the hyp params\ncbc_3.fit(train_data, # instead of X_train, y_train\n          eval_set=valid_data, # instead of (X_valid, y_valid)\n          use_best_model=True, \n          plot=True\n         );\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.95053Z","iopub.status.idle":"2025-01-01T01:27:14.951001Z","shell.execute_reply.started":"2025-01-01T01:27:14.950756Z","shell.execute_reply":"2025-01-01T01:27:14.95078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nscore = cbc_3.get_best_score()#['validation_0']['AUC']\nprint(score)\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.952779Z","iopub.status.idle":"2025-01-01T01:27:14.953275Z","shell.execute_reply.started":"2025-01-01T01:27:14.953004Z","shell.execute_reply":"2025-01-01T01:27:14.953045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#cbc_3.get_all_params()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.954598Z","iopub.status.idle":"2025-01-01T01:27:14.955081Z","shell.execute_reply.started":"2025-01-01T01:27:14.954818Z","shell.execute_reply":"2025-01-01T01:27:14.954841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n%%time\n\nfrom sklearn.model_selection import StratifiedKFold\n\nn_fold = 5 # amount of data folds\nfolds = StratifiedKFold(n_splits=n_fold, shuffle=True, random_state=42)\n\ncbc_4 = CatBoostClassifier(\n        loss_function='Logloss',\n\n        auto_class_weights='SqrtBalanced', #'Balanced'or 'SqrtBalanced' These  give same results as my calcs\n        eval_metric='AUC',\n\n        iterations=100,\n\n        verbose=False,\n        #task_type='GPU',\n        #border_count= 32, # a recommendation to speed up the process\n        early_stopping_rounds=100,              \n        random_seed=42,)\n\ntest_data = Pool(data=dropped_test_df,\n                 cat_features=categorical_features_indices)\n\nscores = []\nprediction = np.zeros(dropped_test_df.shape[0])\nfor fold_n, (train_index, test_index) in enumerate(folds.split(X, Y)):\n    \n    X_train, X_test = X.iloc[train_index], X.iloc[test_index] # train and validation data splits\n    y_train, y_test = Y[train_index], Y[test_index]\n    \n    train_data = Pool(data=X_train, \n                      label=y_train,\n                      cat_features=categorical_features_indices)\n    valid_data = Pool(data=X_test, \n                      label=y_test,\n                      cat_features=categorical_features_indices)\n    \n\n    cbc_4.fit(train_data,\n              eval_set=valid_data, \n              use_best_model=True\n             )\n    \n    score = cbc_4.get_best_score()['validation']['AUC']\n    print(score)\n    scores.append(score)\n\n    y_pred = cbc_4.predict_proba(test_data)[:, 1]\n    prediction += y_pred\n\nprediction /= n_fold\nprint('CV mean: {:.4f}, CV std: {:.4f}'.format(np.mean(scores), np.std(scores)))\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.957211Z","iopub.status.idle":"2025-01-01T01:27:14.957692Z","shell.execute_reply.started":"2025-01-01T01:27:14.957446Z","shell.execute_reply":"2025-01-01T01:27:14.95747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nwith open(pickle_path+'cbc_4.pickle', 'wb') as to_write:\n    pickle.dump(cbc_4_model, to_write)\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.959134Z","iopub.status.idle":"2025-01-01T01:27:14.95961Z","shell.execute_reply.started":"2025-01-01T01:27:14.959357Z","shell.execute_reply":"2025-01-01T01:27:14.95938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##################DOING THIS DID NOT DO BETTER THAN i WAS DOING BEFORE#####################################","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.960794Z","iopub.status.idle":"2025-01-01T01:27:14.961282Z","shell.execute_reply.started":"2025-01-01T01:27:14.961019Z","shell.execute_reply":"2025-01-01T01:27:14.961062Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#################################################END#######################################################","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.962802Z","iopub.status.idle":"2025-01-01T01:27:14.963299Z","shell.execute_reply.started":"2025-01-01T01:27:14.963049Z","shell.execute_reply":"2025-01-01T01:27:14.963073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##########################START###############################","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.96429Z","iopub.status.idle":"2025-01-01T01:27:14.964763Z","shell.execute_reply.started":"2025-01-01T01:27:14.964523Z","shell.execute_reply":"2025-01-01T01:27:14.964545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n%%time\n###########the one to run\ncbc_base = CatBoostRegressor(\n                        loss_function='RMSE',\n                        cat_features=categorical_features_indices,\n                        eval_metric='RMSE',# \n                        task_type='GPU',\n                        gpu_ram_part=0.9,\n                        border_count= 50, \n                        iterations=1000, #def=1000\n                        boosting_type='Plain', #for large dataset, or Ordered for smaller ones\n                        verbose=False,\n                        #nan_mode='Min',\n                        #bootstrap_type= 'Bernoulli', \n                        #subsample=0.9,\n                        #nan_mode='Min'\n                        random_state=42)\n#eval_set = [(X_test, y_test)]\ncbc_base.fit(X_train , y_train, verbose=0)#, eval_set=eval_set, early_stopping_rounds=10,verbose=0)\ntest_predictions = cbc_base.predict(X_test)\n\n#orig_sz_pred = np.expm1(test_predictions) #do at end?\ntest_rmsle = rmsle_score(y_test, test_predictions)#[:,1])\n\n\nprint(f\"Base RMSLE score: {np.round(test_rmsle,6)}\") \n\nwith open(pickle_path+'cbc_base.pickle', 'wb') as to_write:\n    pickle.dump(cbc_base, to_write)\n    \n\n#Base RMSLE score: 1.050956\n#CPU times: user 2min 35s, sys: 5.81 s, total: 2min 41s\n#Wall time: 2min 1s","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.966559Z","iopub.status.idle":"2025-01-01T01:27:14.967048Z","shell.execute_reply.started":"2025-01-01T01:27:14.966784Z","shell.execute_reply":"2025-01-01T01:27:14.966807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.968665Z","iopub.status.idle":"2025-01-01T01:27:14.969152Z","shell.execute_reply.started":"2025-01-01T01:27:14.96889Z","shell.execute_reply":"2025-01-01T01:27:14.968914Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#cbc_base.get_all_params()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.9712Z","iopub.status.idle":"2025-01-01T01:27:14.971675Z","shell.execute_reply.started":"2025-01-01T01:27:14.971432Z","shell.execute_reply":"2025-01-01T01:27:14.971456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#stop\n#Y.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.972788Z","iopub.status.idle":"2025-01-01T01:27:14.973278Z","shell.execute_reply.started":"2025-01-01T01:27:14.973007Z","shell.execute_reply":"2025-01-01T01:27:14.973048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n%%time\n\n\ncbc0 = CatBoostRegressor(\n        loss_function='RMSE',\n        cat_features=categorical_features_indices,\n        #eval_metric='AUC',#https://catboost.ai/en/docs/concepts/loss-functions-classification\n        eval_metric='RMSE', #could try 'Logloss'?\n        task_type='GPU',\n        gpu_ram_part=0.9,\n        border_count= 50,#26,\n        iterations=1100,#3000, #1000def\n        #learning_rate=0.03, #was 0.07 #def 0.03 am I reading the specs right? \n        #if I dont set this directly default is set based on db params and num of iterations set???\n        boosting_type='Plain', #for large dataset, or Ordered for smaller ones\n        random_state=42,\n        verbose=False,      \n        #scale_pos_weight=1.04, #or sqrt 2.67 #################\n        #use one or the other\n        #auto_class_weights='SqrtBalanced', #'Balanced'or 'SqrtBalanced'\n        \n\n        depth=7,#6,\n        min_data_in_leaf=1,#def=1 int\n\n        bootstrap_type= 'Bayesian',\n        bagging_temperature= 0.1,\n\n        #next is a param to keep mem usage down\n        #max_ctr_complexity=2 #def=4\n        #bootstrap_type= 'Bernoulli', \n        #subsample=0.9,\n        \n        #this is the mechanism for early stopping in cbc\n        #od_type = \"Iter\",\n        #od_wait = 100,        \n        #early_stopping_rounds=100, #if no improvement after 100 rounds stop\n        #use_best_model=True,############trying to fix oomem issues have to do this with pools and eval sets\n    ####################################################################\n        #nan_mode='Min',#or 'Max'Type of base_dict_xgbc['xgbc0'] before the function call: <class 'str'>\n        \n        grow_policy='Lossguide',  # can only use max leaves if loss_guide is the policy\n        max_leaves=41, #def=31\n        l2_leaf_reg = 0.1,\n        random_strength=0.01, #0.1, #def 1\n\n\n    #########################################################\n        )\n\n#cbc0.fit(X_train, y_train)#, eval_set=(X_test,y_test))\n#cbc0.fit(X, Y)\n#When you set is_unbalance: True, the algorithm will try to Automatically balance the weight of the dominated label OR\n#sample_pos_weight = number of negative samples / number of positive samples scale_pos_weight=3.7259 based on churn vs not intrain set\n#cbc0_params=cbc0.get_all_params()\n\n#cbc0_params = get_default_params(cbc0, True)\nbase_dict_cbc = {}\nbase_dict_cbc['cbc0'], cbc0_best_params = run_base_model(cbc0, True)\n\nwith open(pickle_path+'cbc_clf0.pickle', 'wb') as to_write:\n    pickle.dump(cbc0, to_write)\n\nprint(f\"Benchmark RMSLE score: {np.round(base_dict_cbc['cbc0']['rmsle'],6)}\")\nprint(f\"Benchmark CV_MEAN_RMSLE score: {np.round(base_dict_cbc['cbc0']['std_cv'],6)}\")\nprint(f\"Benchmark CV_MEAN score: {np.round(base_dict_cbc['cbc0']['mean_cv'],6)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.975412Z","iopub.status.idle":"2025-01-01T01:27:14.9759Z","shell.execute_reply.started":"2025-01-01T01:27:14.975654Z","shell.execute_reply":"2025-01-01T01:27:14.975678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n\nMean CV score: -1.0479301781313755\nThe test RMSLE score: 1.0500149145205684\nBenchmark RMSLE score: 1.050015\nBenchmark CV_MEAN_RMSLE score: 0.004064\nBenchmark CV_MEAN score: -1.04793\nCPU times: user 5min 14s, sys: 1min 2s, total: 6min 16s\nWall time: 3min 18s\n#1100\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.977222Z","iopub.status.idle":"2025-01-01T01:27:14.977561Z","shell.execute_reply.started":"2025-01-01T01:27:14.977397Z","shell.execute_reply":"2025-01-01T01:27:14.977417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nprint(base_dict_cbc['cbc0']['cv_results'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.978732Z","iopub.status.idle":"2025-01-01T01:27:14.979056Z","shell.execute_reply.started":"2025-01-01T01:27:14.978884Z","shell.execute_reply":"2025-01-01T01:27:14.978898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n#from sklearn.base import clone\nrandom_states = [42, 123, 456, 789, 1213]\nthe_res_c, the_best_rmsle_c, the_rand_st_c= run_base_model_many_splits(cbc0, X, Y, 0.25, random_states)\n\nprint(the_best_rmsle_c)\nprint(the_rand_st_c)\nprint(the_res_c)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.980502Z","iopub.status.idle":"2025-01-01T01:27:14.980807Z","shell.execute_reply.started":"2025-01-01T01:27:14.980657Z","shell.execute_reply":"2025-01-01T01:27:14.980672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_importance_df= cbc0.get_feature_importance(prettified=True)\nfeature_importance_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.982228Z","iopub.status.idle":"2025-01-01T01:27:14.98254Z","shell.execute_reply.started":"2025-01-01T01:27:14.982388Z","shell.execute_reply":"2025-01-01T01:27:14.982403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#from matplotlib import pyplot as plt\n#import seaborn as sns\n\nplt.figure(figsize=(12, 6));\nsns.barplot(x=\"Importances\", y=\"Feature Id\", data=feature_importance_df);\nplt.title('CatBoost features importance:');\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.98389Z","iopub.status.idle":"2025-01-01T01:27:14.984239Z","shell.execute_reply.started":"2025-01-01T01:27:14.984076Z","shell.execute_reply":"2025-01-01T01:27:14.984094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#fix this: either start with a pool or  what is:\"ImportError: cannot import name '_fit_context' from 'sklearn.base' (/opt/conda/lib/python3.10/site-packages/sklearn/base.py)\"\n'''\nimport shap\nexplainer = shap.TreeExplainer(cbc0) # insert your model\n#shap_values = explainer.shap_values(train_data) # insert your train Pool object\nshap_values = explainer.shap_values(X_train) # insert your train Pool object\nshap.initjs()\nshap.force_plot(explainer.expected_value, shap_values[:100,:], X_train.iloc[:100,:])\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.985001Z","iopub.status.idle":"2025-01-01T01:27:14.985357Z","shell.execute_reply.started":"2025-01-01T01:27:14.985197Z","shell.execute_reply":"2025-01-01T01:27:14.985214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#shap.summary_plot(shap_values, X_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.986924Z","iopub.status.idle":"2025-01-01T01:27:14.987277Z","shell.execute_reply.started":"2025-01-01T01:27:14.987103Z","shell.execute_reply":"2025-01-01T01:27:14.987119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n#gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.988058Z","iopub.status.idle":"2025-01-01T01:27:14.988373Z","shell.execute_reply.started":"2025-01-01T01:27:14.988214Z","shell.execute_reply":"2025-01-01T01:27:14.988229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cbc0_params = cbc0.get_all_params()\ncbc0_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.989623Z","iopub.status.idle":"2025-01-01T01:27:14.989961Z","shell.execute_reply.started":"2025-01-01T01:27:14.989792Z","shell.execute_reply":"2025-01-01T01:27:14.989809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ncbc_param_grid = {\n          \n            #'depth': [7, 8], #gpu max 8 unless logless, cpu int to 16\n            #6 (16 if the growing policy is set to Lossguide)\n            #'min_data_in_leaf':[1,2],#def=1 int\n            #'depth':range(5,10,2),\n            #'min_data_in_leaf':range(1,6,2)#def=1 int\n            \n            #'l2_leaf_reg': [0.1, 0.5,], #def=3 fl\n            #'l2_leaf_reg':range(1,6,2),  #def=3,fl any pos val\n            #'grow_policy':['Lossguide'],\n            #'border_count':[10, 15, 20, 25, 30, 35, 40, 45, 50],#def 256, > better result , < quicker training\n            #'bagging_temperature':[0.15, 0.2, 0.25] #def=1,, fl 0 ->\n\n            \n            #'learning_rate': [0.01,0.02,0.03, 0.04, 0.05], #def 0.03\n            #'iterations':[3000, 5000],\n            #'random_strength':[0.01, 0.2, 0.3,], #def=1, fl try it, sounds like just a seed for randomness?\n\n            #'max_leaves':range(70,91,10), #def=31 #60 worked well\n            #'bootstrap_type':['Bernoulli'], #not bayesian if I want to tune subsample and colsample_by_level\n            #'bootstrap_type':['Bayesian'], #not bayesian if I want to tune subsample and colsample_by_level\n            #'subsample':[0.4,0.5,0.6,0.7,0.8,0.9,1], #fl mainly 1 and can not set if bootstrap type = bayesian\n            #'colsample_bylevel':[0.7,0.8,0.9], #def1\n            #'scale_pos_weight': [1, 2.25, 5],\n            #'max_ctr_complexity':[1,2,3,4,5],\n            #'bagging_temperature':[0.01, 0.1, 2, 10] #def=1,, fl 0 ->\n            #'leaf_estimation_iterations': range(5,16,5), #def for classifier =10        \n            #'use_best_model': ['True'],\n            #'logging_level':['Silent'],\n            #'random_seed': [42]\n            'grow_policy':['Lossguide'],# def=SymmetricTree\n            'max_leaves':[41, 91]#def 31, only if lossguide\n    }\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.99125Z","iopub.status.idle":"2025-01-01T01:27:14.991582Z","shell.execute_reply.started":"2025-01-01T01:27:14.991422Z","shell.execute_reply":"2025-01-01T01:27:14.991439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#######################################################################################################################","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.992908Z","iopub.status.idle":"2025-01-01T01:27:14.993245Z","shell.execute_reply.started":"2025-01-01T01:27:14.993086Z","shell.execute_reply":"2025-01-01T01:27:14.993103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#clf0_model = pickle.load(open('/kaggle/working/cbc_clf0.pickle', 'rb'))\n#base_model = pickle.load(open('/kaggle/working/cbc_base.pickle', 'rb'))\n#cbc0_params = get_default_params(base_model, True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.994921Z","iopub.status.idle":"2025-01-01T01:27:14.995266Z","shell.execute_reply.started":"2025-01-01T01:27:14.995104Z","shell.execute_reply":"2025-01-01T01:27:14.995121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#cbc0_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.99645Z","iopub.status.idle":"2025-01-01T01:27:14.99677Z","shell.execute_reply.started":"2025-01-01T01:27:14.996615Z","shell.execute_reply":"2025-01-01T01:27:14.996631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n%%time\n\ncbc_model = cbc0\n#trial_params= deepcopy(cbc_base_params)\n#trial_model =  CatBoostClassifier(**trial_params)\n#trial_model= cbc_base#cbc_base\ngrid_dict_cbc, grid_model = get_grid_cv(cbc_param_grid, cbc_model, 'cbc_grid', True)\n\ncombo_dict_cbc = base_dict_cbc | grid_dict_cbc\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.998188Z","iopub.status.idle":"2025-01-01T01:27:14.998672Z","shell.execute_reply.started":"2025-01-01T01:27:14.998426Z","shell.execute_reply":"2025-01-01T01:27:14.998451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#print_grid_score(grid_count, grid_dict_cbc, 'cbc_grid_', MY_SCORE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:14.999687Z","iopub.status.idle":"2025-01-01T01:27:15.000182Z","shell.execute_reply.started":"2025-01-01T01:27:14.999913Z","shell.execute_reply":"2025-01-01T01:27:14.999937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#grid_model = pickle.load(open('/kaggle/working/cbc_grid_1.pickle', 'rb'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.002264Z","iopub.status.idle":"2025-01-01T01:27:15.002753Z","shell.execute_reply.started":"2025-01-01T01:27:15.002505Z","shell.execute_reply":"2025-01-01T01:27:15.00253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grid_model.best_params_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.003776Z","iopub.status.idle":"2025-01-01T01:27:15.004122Z","shell.execute_reply.started":"2025-01-01T01:27:15.003937Z","shell.execute_reply":"2025-01-01T01:27:15.003953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grid_model.best_score_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.005049Z","iopub.status.idle":"2025-01-01T01:27:15.005361Z","shell.execute_reply.started":"2025-01-01T01:27:15.005209Z","shell.execute_reply":"2025-01-01T01:27:15.005225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"-1.0479071125280002 w 41","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.00641Z","iopub.status.idle":"2025-01-01T01:27:15.006742Z","shell.execute_reply.started":"2025-01-01T01:27:15.006575Z","shell.execute_reply":"2025-01-01T01:27:15.006591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grid_model.cv_results_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.008455Z","iopub.status.idle":"2025-01-01T01:27:15.00877Z","shell.execute_reply.started":"2025-01-01T01:27:15.00861Z","shell.execute_reply":"2025-01-01T01:27:15.008626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n%%time\n'''\ndefault_params_cbc = {}\n\nfor key in cbc0_params.keys():\n    default_params_cbc[key] = cbc0_params[key][0]\n\n'''\n#cbc_rs_model = CatBoostClassifier(**default_params_cbc)\ncbc_rs_model = cbc0\nrs_dict_cbc, cbc_rs_best_params, cbc_rs_model = get_random_searchcv(cbc_param_grid, cbc_rs_model, 'cbc_rs')\n\n#results_dict_cbc = combo_dict_cbc | rs_dict_cbc\n\n\nwith open(pickle_path+'cbc_rs.pickle', 'wb') as to_write:\n    pickle.dump(cbc_rs_model, to_write)\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.009891Z","iopub.status.idle":"2025-01-01T01:27:15.010236Z","shell.execute_reply.started":"2025-01-01T01:27:15.010072Z","shell.execute_reply":"2025-01-01T01:27:15.010089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cbc_rs_model.best_params_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.011414Z","iopub.status.idle":"2025-01-01T01:27:15.011752Z","shell.execute_reply.started":"2025-01-01T01:27:15.011583Z","shell.execute_reply":"2025-01-01T01:27:15.0116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ncbc_rs_model.best_score_\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.012816Z","iopub.status.idle":"2025-01-01T01:27:15.013176Z","shell.execute_reply.started":"2025-01-01T01:27:15.012982Z","shell.execute_reply":"2025-01-01T01:27:15.012999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cbc_rs_model.cv_results_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.014137Z","iopub.status.idle":"2025-01-01T01:27:15.014465Z","shell.execute_reply.started":"2025-01-01T01:27:15.014302Z","shell.execute_reply":"2025-01-01T01:27:15.014319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n\nprint(f\"Benchmark AUC score: {np.round(results_dict_cbc['cbc0']['test roc auc score'],4)}\")\nprint_grid_score(grid_count, results_dict_cbc, 'cbc_grid_', MY_SCORE)\nprint(f\"Randomized search AUC score: {np.round(results_dict_cbc['cbc_rs_']['test roc auc score'],4)}\")\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.015733Z","iopub.status.idle":"2025-01-01T01:27:15.01611Z","shell.execute_reply.started":"2025-01-01T01:27:15.015886Z","shell.execute_reply":"2025-01-01T01:27:15.015901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.017161Z","iopub.status.idle":"2025-01-01T01:27:15.017472Z","shell.execute_reply.started":"2025-01-01T01:27:15.017315Z","shell.execute_reply":"2025-01-01T01:27:15.017331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#SHAP values \n#model = cbc_rs_model.best_estimator_\n\n#import shap\n'''\nexplainer = shap.Explainer(model)\nshap_values = explainer(X_test)\nshap_importance = shap_values.abs.mean(0).values\nsorted_idx = shap_importance.argsort()\nfig = plt.figure(figsize=(12, 6))\nplt.barh(range(len(sorted_idx)), shap_importance[sorted_idx], align='center')\nplt.yticks(range(len(sorted_idx)), np.array(X_test.columns)[sorted_idx])\nplt.title('SHAP Importance')\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.018927Z","iopub.status.idle":"2025-01-01T01:27:15.019267Z","shell.execute_reply.started":"2025-01-01T01:27:15.019108Z","shell.execute_reply":"2025-01-01T01:27:15.019124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n#shap.plots.bar(shap_values, max_display=X_test.shape[0])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.02032Z","iopub.status.idle":"2025-01-01T01:27:15.020663Z","shell.execute_reply.started":"2025-01-01T01:27:15.0205Z","shell.execute_reply":"2025-01-01T01:27:15.020517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#alternative to SHAP\n'''\nimport sklearn.inspection.permutation_importance\n\nmodel = cbc0.best_estimator_\n\nperm_importance = permutation_importance(cbc0, X_test, y_test, n_repeats=10, random_state=42)\nsorted_idx = perm_importance.importances_mean.argsort()\nfig = plt.figure(figsize=(12, 6))\nplt.barh(range(len(sorted_idx)), perm_importance.importances_mean[sorted_idx], align='center')\nplt.yticks(range(len(sorted_idx)), np.array(X_test.columns)[sorted_idx])\nplt.title('Permutation Importance')\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.022083Z","iopub.status.idle":"2025-01-01T01:27:15.022416Z","shell.execute_reply.started":"2025-01-01T01:27:15.022253Z","shell.execute_reply":"2025-01-01T01:27:15.02227Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **lightGBM**","metadata":{}},{"cell_type":"code","source":"#stop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.023237Z","iopub.status.idle":"2025-01-01T01:27:15.023552Z","shell.execute_reply.started":"2025-01-01T01:27:15.023401Z","shell.execute_reply":"2025-01-01T01:27:15.023417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n#LightGBMError: bin size 16734 cannot run on GPU seems to be an error if categories contain > 256 bins?\nlgbc_base = LGBMRegressor(\n                        #device_type='gpu', #or 'gpu' generic\n                        objective='regression_l2',\n                        metric='rmse',\n                        #look in docs for changing the behaviour of this\n                        categorical_feature=categorical_features_indices,\n                        #n_estimators=100,\n                        #max_bin=290, #vintage has 290 unique entries\n                        max_bin=50, #slightly better 8873 vs 8872\n                        random_state=42)\n\neval_set = [(X_test, y_test)]\nlgbc_base.fit(X_train , y_train,verbose=0)\ntest_predictions = lgbc_base.predict(X_test)\ntest_rmsle = rmsle_score(y_test, test_predictions)#[:,1])\n\n\nprint(f\"Base RMSLE score: {np.round(test_rmsle,6)}\") \n\n#Base RMSLE score: 1.052561\n#CPU times: user 17.7 s, sys: 153 ms, total: 17.9 s\n#Wall time: 4.83 s","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.025422Z","iopub.status.idle":"2025-01-01T01:27:15.025907Z","shell.execute_reply.started":"2025-01-01T01:27:15.025649Z","shell.execute_reply":"2025-01-01T01:27:15.025684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nBase RMSLE score: 1.052561\nCPU times: user 23 s, sys: 832 ms, total: 23.8 s\nWall time: 7.48 s\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.02728Z","iopub.status.idle":"2025-01-01T01:27:15.027759Z","shell.execute_reply.started":"2025-01-01T01:27:15.027504Z","shell.execute_reply":"2025-01-01T01:27:15.027528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.029075Z","iopub.status.idle":"2025-01-01T01:27:15.029408Z","shell.execute_reply.started":"2025-01-01T01:27:15.029246Z","shell.execute_reply":"2025-01-01T01:27:15.029262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n#I think that in my grid method if I have no parameter set in below I pull from above, should double check that one of these days\nlgbc0 = LGBMRegressor(\n                        #device_type='gpu',#'cuda', #or 'gpu' generic\n                        objective='regression_l2',\n                        categorical_feature=categorical_features_indices, #same as catboost give the indices of cols, also tyhey must first be cast to int32\n                        metric='rmse',\n                        #n_estimators= 2000,  #def=100 \n                        #learning_rate=0.01,#def 0.1,\n                      \n                        #boosting='dart',\n                        max_bin=50,#15,#26, #or 290 or 152\n                        random_state=42,\n                        #scale_pos_weight=1.05, \n    \n                        n_estimators= 5000,  #def=100 \n                        learning_rate=0.005,#def 0.1,                          \n                        \n                        colsample_bytree=1,#def=1,\n                        subsample=1, # #def =1,\n                        max_depth=-1, #def =-1 ie none\n                        num_leaves=31,#def 31 \n                        min_data_in_leaf= 7,\n                        min_child_weight=1e-8, #didnt seem to make diff   def=1e-3 \n                        ##min_child_samples=160,\n                        bagging_freq=5, #doesnt seem to matter\n                        reg_alpha= 1e-8, #def 0.0,  \n                        reg_lambda= 0.32, # def0.0\n\n                        #early_stopping_round= 20 #0\n                        #'early_stopping_min_delta':  # the amount of improvent to see if esr(above)>0,\n                        #early stopping metric to improve by at least this delta to be considered an improvement\n\n                        )\n#lgbc0.fit(X_train, y_train)\n\n#When you set is_unbalance: True, the algorithm will try to Automatically balance the weight of the dominated label OR\n#sample_pos_weight = number of negative samples / number of positive samples scale_pos_weight=3.7259 based on churn vs not intrain set\n#scoe mcc before ********** Benchmark MCC score: 0.984798\n#lgbc0_params = get_default_params(lgbc0)\nbase_dict_lgbc = {}\nbase_dict_lgbc['lgbc0'], lgbc0_best_params = run_base_model(lgbc0)\n\nwith open(pickle_path+'lgbc_clf0.pickle', 'wb') as to_write:\n    pickle.dump(lgbc0, to_write)  \n\n\nprint(f\"Benchmark RMSLE score: {np.round(base_dict_lgbc['lgbc0']['rmsle'],6)}\")\nprint(f\"Benchmark CV_MEAN_RMSLE score: {np.round(base_dict_lgbc['lgbc0']['std_cv'],6)}\")\nprint(f\"Benchmark CV_MEAN score: {np.round(base_dict_lgbc['lgbc0']['mean_cv'],6)}\")\n#Benchmark RMSLE score: 1.051251\n#Benchmark CV_MEAN_RMSLE score: 0.002737\n#Benchmark CV_MEAN score: -1.049946\n#CPU times: user 56min 51s, sys: 2min 5s, total: 58min 57s\n#Wall time: 19min 32s\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.031014Z","iopub.status.idle":"2025-01-01T01:27:15.031375Z","shell.execute_reply.started":"2025-01-01T01:27:15.031214Z","shell.execute_reply":"2025-01-01T01:27:15.031232Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nThe test RMSLE score: 1.0523495625596264\nBenchmark RMSLE score: 1.05235\nBenchmark CV_MEAN_RMSLE score: 0.004102\nBenchmark CV_MEAN score: -1.050447\nCPU times: user 1h 44min 49s, sys: 3min 51s, total: 1h 48min 40s\nWall time: 34min 4s\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.03285Z","iopub.status.idle":"2025-01-01T01:27:15.033377Z","shell.execute_reply.started":"2025-01-01T01:27:15.033108Z","shell.execute_reply":"2025-01-01T01:27:15.033133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lightgbm_pred = lgbc0.predict(X_test)\n\nens_rmsle = rmsle_score(y_test, lightgbm_pred)\nprint(f\"Base RMSLE score: {np.round(ens_rmsle,6)}\") ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.035288Z","iopub.status.idle":"2025-01-01T01:27:15.035764Z","shell.execute_reply.started":"2025-01-01T01:27:15.035509Z","shell.execute_reply":"2025-01-01T01:27:15.035533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nThe test RMSLE score: 1.0525584891341484\nBenchmark RMSLE score: 1.052558\nBenchmark CV_MEAN_RMSLE score: 0.004128\nBenchmark CV_MEAN score: -1.050584\nCPU times: user 2h 7min 55s, sys: 18.4 s, total: 2h 8min 13s\nWall time: 32min 53s\n'''\n#w\n'''\n                        n_estimators= 5000,  #def=100 \n                        learning_rate=0.005,#def 0.1,                          \n                        \n                        colsample_bytree=0.9,#0.5,\n                        subsample=0.9, # #0.6,\n                        #max_depth=60, #def =-1 ie none\n                        num_leaves=41,#def 31 \n                        min_data_in_leaf= 10,\n                        min_child_weight=0.2, #coulkd be 1\n                        #or\n                        #min_child_samples=160,\n                        bagging_freq=12, #was7\n                        reg_alpha= 5e-4, #def 0.0,  \n                        reg_lambda= 5e-4, # def0.0\n\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.037802Z","iopub.status.idle":"2025-01-01T01:27:15.038173Z","shell.execute_reply.started":"2025-01-01T01:27:15.037974Z","shell.execute_reply":"2025-01-01T01:27:15.037991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n#from sklearn.base import clone\nrandom_states = [42, 123, 456, 789, 1213]\nthe_res, the_best_rmsle, the_rand_st= run_base_model_many_splits(lgbc0, X, Y, 0.25, random_states)\n\nprint(the_best_rmsle)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.039308Z","iopub.status.idle":"2025-01-01T01:27:15.039624Z","shell.execute_reply.started":"2025-01-01T01:27:15.039471Z","shell.execute_reply":"2025-01-01T01:27:15.039487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(the_rand_st)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.040414Z","iopub.status.idle":"2025-01-01T01:27:15.040787Z","shell.execute_reply.started":"2025-01-01T01:27:15.040624Z","shell.execute_reply":"2025-01-01T01:27:15.040641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(the_res)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.041932Z","iopub.status.idle":"2025-01-01T01:27:15.042278Z","shell.execute_reply.started":"2025-01-01T01:27:15.042111Z","shell.execute_reply":"2025-01-01T01:27:15.042127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(base_dict_lgbc['lgbc0']['cv_results'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.043091Z","iopub.status.idle":"2025-01-01T01:27:15.043423Z","shell.execute_reply.started":"2025-01-01T01:27:15.043258Z","shell.execute_reply":"2025-01-01T01:27:15.043274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.044712Z","iopub.status.idle":"2025-01-01T01:27:15.045016Z","shell.execute_reply.started":"2025-01-01T01:27:15.044869Z","shell.execute_reply":"2025-01-01T01:27:15.044884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the feature importances\nimportance = lgbc0.feature_importances_\nfeature_names = X_train.columns\n\n# Create a DataFrame with the feature importances\nfeature_importance_df = pd.DataFrame({'Feature': feature_names, 'Importance': importance})\n\n# Sort the DataFrame by importance\nfeature_importance_df = feature_importance_df.sort_values(by='Importance', ascending=False)\n\n# Plot using seaborn\nplt.figure(figsize=(10, 8))\nsns.barplot(x='Importance', y='Feature', data=feature_importance_df)\nplt.title(\"Feature Importance\")\nplt.grid(False)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.046282Z","iopub.status.idle":"2025-01-01T01:27:15.046659Z","shell.execute_reply.started":"2025-01-01T01:27:15.046453Z","shell.execute_reply":"2025-01-01T01:27:15.046474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_importance_df.head(20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.047486Z","iopub.status.idle":"2025-01-01T01:27:15.047885Z","shell.execute_reply.started":"2025-01-01T01:27:15.047671Z","shell.execute_reply":"2025-01-01T01:27:15.047688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbc_param_grid = {\n            #'reg_alpha':[1e-8,0 ],#def=0.0\n            #'reg_lambda':[0.35, 0.3, 0.25], #def=0.0\n\n            #'num_leaves': [31], #31\n            #'min_data_in_leaf':[6,7,8,],#def=20\n            #'min_child_weight':[1e-8, 1e-5, 1e-3, 0.1, 1, 3]#def 1e-3\n            \n            #'max_depth':[-1] #def -1 ... none\n            #'n_estimators': [ 5000],  #def=100 \n            'learning_rate':[0.001, 0.005, 0.0005,0.0001],#def 0.1,\n            #'max_bin':[365, 375, 395],\n            #'colsample_bytree':[ 0.4, 0.5, 0.9, 1],\n            #'subsample':[0.4, 0.5, 0.9, 1],\n    \n            #'bagging_freq': [3,4,5, 7 ], #def=0\n\n    }\n  #{'subsample': 0.5, 'colsample_bytree': 0.7, 'bagging_freq': 5},\n  #{'subsample': 1, 'colsample_bytree': 0.4, 'bagging_freq': 10},","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.050126Z","iopub.status.idle":"2025-01-01T01:27:15.050607Z","shell.execute_reply.started":"2025-01-01T01:27:15.050359Z","shell.execute_reply":"2025-01-01T01:27:15.050383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n%%time\n\nlgbc_model = lgbc0\n\ngrid_dict_lgbc, grid_model = get_grid_cv(lgbc_param_grid, lgbc_model, 'lgbc_grid')\n\ncombo_dict_lgbc = base_dict_lgbc | grid_dict_lgbc\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.051692Z","iopub.status.idle":"2025-01-01T01:27:15.052188Z","shell.execute_reply.started":"2025-01-01T01:27:15.051915Z","shell.execute_reply":"2025-01-01T01:27:15.051938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#print_grid_score(grid_count, grid_dict_lgbc, 'lgbc_grid_', MY_SCORE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.053989Z","iopub.status.idle":"2025-01-01T01:27:15.054477Z","shell.execute_reply.started":"2025-01-01T01:27:15.054233Z","shell.execute_reply":"2025-01-01T01:27:15.054256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n#grid_model = pickle.load(open('/kaggle/working/lgbc_grid_1.pickle', 'rb'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.055697Z","iopub.status.idle":"2025-01-01T01:27:15.056208Z","shell.execute_reply.started":"2025-01-01T01:27:15.055939Z","shell.execute_reply":"2025-01-01T01:27:15.055962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#grid_model.best_estimator_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.057531Z","iopub.status.idle":"2025-01-01T01:27:15.058016Z","shell.execute_reply.started":"2025-01-01T01:27:15.057764Z","shell.execute_reply":"2025-01-01T01:27:15.057788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grid_model.best_params_\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.060232Z","iopub.status.idle":"2025-01-01T01:27:15.06074Z","shell.execute_reply.started":"2025-01-01T01:27:15.060473Z","shell.execute_reply":"2025-01-01T01:27:15.060497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grid_model.best_score_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.062175Z","iopub.status.idle":"2025-01-01T01:27:15.062516Z","shell.execute_reply.started":"2025-01-01T01:27:15.062341Z","shell.execute_reply":"2025-01-01T01:27:15.062357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grid_model.cv_results_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.063351Z","iopub.status.idle":"2025-01-01T01:27:15.06368Z","shell.execute_reply.started":"2025-01-01T01:27:15.063525Z","shell.execute_reply":"2025-01-01T01:27:15.063541Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#stop","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.064934Z","iopub.status.idle":"2025-01-01T01:27:15.065285Z","shell.execute_reply.started":"2025-01-01T01:27:15.065115Z","shell.execute_reply":"2025-01-01T01:27:15.065132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nlgbc_rs_model = lgbc0\n\nrs_dict_lgbc, lgbc_rs_best_params, lgbc_rs_model = get_random_searchcv(lgbc_param_grid, lgbc_rs_model, 'lgbc_rs')\n\n#results_dict_lgbc = combo_dict_lgbc | rs_dict_lgbc\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.066521Z","iopub.status.idle":"2025-01-01T01:27:15.066834Z","shell.execute_reply.started":"2025-01-01T01:27:15.066682Z","shell.execute_reply":"2025-01-01T01:27:15.066697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nwith open(pickle_path+'lgbc_rs.pickle', 'wb') as to_write:\n    pickle.dump(lgbc_rs_model, to_write)\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.068176Z","iopub.status.idle":"2025-01-01T01:27:15.068529Z","shell.execute_reply.started":"2025-01-01T01:27:15.068346Z","shell.execute_reply":"2025-01-01T01:27:15.068361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbc_rs_model.best_params_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.069648Z","iopub.status.idle":"2025-01-01T01:27:15.069979Z","shell.execute_reply.started":"2025-01-01T01:27:15.06982Z","shell.execute_reply":"2025-01-01T01:27:15.069837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbc_rs_model.best_score_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.071098Z","iopub.status.idle":"2025-01-01T01:27:15.07161Z","shell.execute_reply.started":"2025-01-01T01:27:15.07144Z","shell.execute_reply":"2025-01-01T01:27:15.071457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbc_rs_model.cv_results_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.073064Z","iopub.status.idle":"2025-01-01T01:27:15.073563Z","shell.execute_reply.started":"2025-01-01T01:27:15.073299Z","shell.execute_reply":"2025-01-01T01:27:15.073323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#results_dict_lgbc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.074793Z","iopub.status.idle":"2025-01-01T01:27:15.075283Z","shell.execute_reply.started":"2025-01-01T01:27:15.075015Z","shell.execute_reply":"2025-01-01T01:27:15.075059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#results_dict = results_dict_lgbc | results_dict_xgbc | results_dict_gbc | results_dict_rfc | results_dict_dtc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.076822Z","iopub.status.idle":"2025-01-01T01:27:15.077314Z","shell.execute_reply.started":"2025-01-01T01:27:15.077067Z","shell.execute_reply":"2025-01-01T01:27:15.077093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#save my results dict for later\n'''\nimport csv\n\n# define the function\ndef dict_to_csv(dictionary, filename):\n    keys = list(dictionary.keys())\n    values = list(dictionary.values())\n\n    with open(filename, 'w', newline='') as csvfile:\n        writer = csv.writer(csvfile)\n        writer.writerow(keys)\n        writer.writerows(zip(*values))\n\n# Convert dictionary to CSV\n#results_dict_xgbc\ndict_to_csv(results_dict_xgbc, '/kaggle/working/results_dict.csv')\n#dict_to_csv(results_dict_cbc, '/kaggle/working/results_dict.csv')\n#dict_to_csv(results_dict, '/kaggle/working/results_dict.csv')\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.078586Z","iopub.status.idle":"2025-01-01T01:27:15.079071Z","shell.execute_reply.started":"2025-01-01T01:27:15.078808Z","shell.execute_reply":"2025-01-01T01:27:15.078831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#############################ensembling ###################################################################","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.080784Z","iopub.status.idle":"2025-01-01T01:27:15.081273Z","shell.execute_reply.started":"2025-01-01T01:27:15.081009Z","shell.execute_reply":"2025-01-01T01:27:15.081052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#cbc0 = pickle.load(open('/kaggle/working/cbc_clf0.pickle', 'rb'))\n#lgbc0 = pickle.load(open('/kaggle/working/lgbc_clf0.pickle', 'rb'))\n#xgbc_model = pickle.load(open('/kaggle/working/xgbc_clf0.pickle', 'rb'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.083116Z","iopub.status.idle":"2025-01-01T01:27:15.083604Z","shell.execute_reply.started":"2025-01-01T01:27:15.083344Z","shell.execute_reply":"2025-01-01T01:27:15.083367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#cbc0.fit(X_train, y_train)\n#lgbc0.fit(X_train, y_train)\n#xgbc0.fit(X_train, y_train)\n#BLENDING\n\ndef ensemble_predictions(weights, catboost_pred, lightgbm_pred, xgboost_pred):\n    return weights[0] * catboost_pred + weights[1] * lightgbm_pred + weights[2] * xgboost_pred\n# Make predictions\ncatboost_pred = cbc0.predict(X_test)\nlightgbm_pred = lgbc0.predict(X_test)\nxgboost_pred = xgbc0.predict(X_test)\n\n#weights = [0.6, 0.3, 0.1] # equal weights\nweights = [1/3, 1/3, 1/3] # equal weights\nensemble_pred = ensemble_predictions(weights, catboost_pred, lightgbm_pred, xgboost_pred)\n#ensemble_pred = (catboost_pred + lightgbm_pred + xgboost_pred) / 3\n\nens_rmsle = rmsle_score(y_test, ensemble_pred)\nprint(f\"Base RMSLE score: {np.round(ens_rmsle,6)}\") \n#\n#Base RMSLE score: 1.050286 1/3  each\n\n#Best weights: [0.6 0.3 0.1], Best MSE: 1.0499825230806317\n#Base RMSLE score: 1.049983","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.084656Z","iopub.status.idle":"2025-01-01T01:27:15.085138Z","shell.execute_reply.started":"2025-01-01T01:27:15.084876Z","shell.execute_reply":"2025-01-01T01:27:15.084899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ncatboost_preds = cbc0.predict(dropped_test_df)\nlightgbm_preds = lgbc0.predict(dropped_test_df)\nxgboost_preds = xgbc0.predict(dropped_test_df)\nweights = [0.9, 0.0, 0.1] # equal weights\nensemble_preds = ensemble_predictions(weights, catboost_preds, lightgbm_preds, xgboost_preds)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.086584Z","iopub.status.idle":"2025-01-01T01:27:15.087071Z","shell.execute_reply.started":"2025-01-01T01:27:15.086808Z","shell.execute_reply":"2025-01-01T01:27:15.086831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nsub_df_ens = pd.DataFrame({\n        \"id\": test_df[\"id\"],\n        TARGET: np.expm1(ensemble_preds)\n    })\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.088406Z","iopub.status.idle":"2025-01-01T01:27:15.088751Z","shell.execute_reply.started":"2025-01-01T01:27:15.088578Z","shell.execute_reply":"2025-01-01T01:27:15.088595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path=\"/kaggle/working/submission_ens_new2.csv\"\n        \nsub_df_ens.to_csv(path, index=False)\nprint(f\"Submission exported to {path}\")\n!head /kaggle/working/submission_ens_new2.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.089887Z","iopub.status.idle":"2025-01-01T01:27:15.090296Z","shell.execute_reply.started":"2025-01-01T01:27:15.090133Z","shell.execute_reply":"2025-01-01T01:27:15.090151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import itertools\n#or lets loop through weights that add to 1 and only go to 1 decimal place\nbest_rmsle = float('inf')\nbest_weights = None\n\nfor weights in itertools.product(np.arange(0, 1.1, 0.1), repeat=3):\n  weights = np.array(weights)\n  if np.isclose(np.sum(weights), 1.0, atol=0.01):  #Allow for small floating-point errors\n      ensemble_pred2 = ensemble_predictions(weights, catboost_pred, lightgbm_pred, xgboost_pred)\n      ens_rmsle = rmsle_score(y_test, ensemble_pred2)\n      print(f\"Weights: {weights}, MSE: {ens_rmsle}\")\n      if ens_rmsle < best_rmsle:\n          best_rmsle = ens_rmsle\n          best_weights = weights\n\nprint(f\"\\nBest weights: {best_weights}, Best MSE: {best_rmsle}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.09147Z","iopub.status.idle":"2025-01-01T01:27:15.091805Z","shell.execute_reply.started":"2025-01-01T01:27:15.091639Z","shell.execute_reply":"2025-01-01T01:27:15.091656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\n#cbc0.fit(X_train, y_train)\n#lgbc0.fit(X_train, y_train)\n#xgbc0.fit(X_train, y_train)\n#BLENDING\n\nstacking_data = np.column_stack((catboost_pred, lightgbm_pred, xgboost_pred))\nstacking_model = LinearRegression()\nstacking_model.fit(X_train, y_train)\n\nstacking_pred = stacking_model.predict(X_test)\nstack_rmsle = rmsle_score(y_test, stacking_pred)\nprint(f\"Base RMSLE score: {np.round(stack_rmsle,6)}\") ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.093047Z","iopub.status.idle":"2025-01-01T01:27:15.093383Z","shell.execute_reply.started":"2025-01-01T01:27:15.093224Z","shell.execute_reply":"2025-01-01T01:27:15.093241Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#cbc_model = pickle.load(open('/kaggle/working/cbc_clf0.pickle', 'rb'))\n#lgbc_model = pickle.load(open('/kaggle/working/lgbc_clf0.pickle', 'rb'))\n#xgbc_model = pickle.load(open('/kaggle/working/xgbc_clf0.pickle', 'rb'))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.094862Z","iopub.status.idle":"2025-01-01T01:27:15.095347Z","shell.execute_reply.started":"2025-01-01T01:27:15.095104Z","shell.execute_reply":"2025-01-01T01:27:15.095127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#first runs no eda except to categorize all and fix numbers for each seperate feature cols nas\nsub0 = pd.read_csv('/kaggle/working/submission_cbc_3.csv') #1st cbc\nsub1 = pd.read_csv('/kaggle/working/submission_lgbc_3.csv') #lgbc\nsub2 = pd.read_csv('/kaggle/working/submission_xgbc_3.csv') #xgbc\nsub3 = pd.read_csv('/kaggle/working/submission_vr_3.csv') #above\nsub4 = pd.read_csv('/kaggle/working/submission_ens_new2.csv') #weights[0.6, 0.3, 0.1]\n\n#should have kept a better notes on these.\nsub5 = pd.read_csv('/kaggle/input/first-runs/sub_1.csv') \nsub6 = pd.read_csv('/kaggle/input/first-runs/sub_2.csv') \nsub7 = pd.read_csv('/kaggle/input/first-runs/sub_3.csv')  \nsub8 = pd.read_csv('/kaggle/input/first-runs/sub_4.csv') \nsub9 = pd.read_csv('/kaggle/input/first-runs/lgbm_3.csv') \nsub10 = pd.read_csv('/kaggle/input/first-runs/mode_4.csv')  \nsub11 = pd.read_csv('/kaggle/input/first-runs/mode_6.csv') \n\nsub12 = pd.read_csv('/kaggle/input/misc-best3/sub_cb0_11.csv') \nsub13 = pd.read_csv('/kaggle/input/misc-best3/sub_cb0_16.csv')  \nsub14 = pd.read_csv('/kaggle/input/misc-best3/sub_lgb0_18.csv') \n#m1; mode4+6+lg3 1.02865\n#m2 mode4+6+ens1 1.02865\n#m3 mode4+6+vr_1 1.02865\n#m4 cb+lg+xg_1s  1.11317\n#m5 lg3+ens1+vr_1 1.07769\n#m6 mode+lg3+vr1 as mean 1.03820\n#m7 mode 4+6 +lg3 mean 1.02903\n#m8 sub3+lg3+mode4 as mean  1.03201\n#m9 mean2 sub3, mode4,6  1.03253\n#m10 vr2 cbc,lgb 2  1.04790\n#m11 ens_new lgbc, cbc 1.04750\n#m12 cbc2  1.04754\n#m13 lgbc2   1.04974\n#meanmiscbestoflast3 1.02925\n#mean9-14+vr3 \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.096454Z","iopub.status.idle":"2025-01-01T01:27:15.096919Z","shell.execute_reply.started":"2025-01-01T01:27:15.096676Z","shell.execute_reply":"2025-01-01T01:27:15.0967Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nsub14 = pd.read_csv('/kaggle/input/best-of-local-cv-subs/Submission_lgbc_21.csv')\nsub15 = pd.read_csv('/kaggle/input/best-of-local-cv-subs/submission_lgbc11.csv')\nsub16 = pd.read_csv('/kaggle/input/best-of-local-cv-subs/submission_cbc_15.csv')\nsub17 = pd.read_csv('/kaggle/input/best-of-local-cv-subs/submission_cbc_11.csv')\nsub18 = pd.read_csv('/kaggle/input/best-of-local-cv-subs/submission_cbc_4.csv')\nsub19 = pd.read_csv('/kaggle/input/best-of-local-cv-subs/submission_xgbc_19.csv')\nsub20 = pd.read_csv('/kaggle/input/best-of-local-cv-subs/submission_xgbc_9.csv')\nsub21 = pd.read_csv('/kaggle/input/best-of-local-cv-subs/submission_xgbc_6.csv')\nsub22 = pd.read_csv('/kaggle/working/submission_mixens.csv')\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.098488Z","iopub.status.idle":"2025-01-01T01:27:15.098832Z","shell.execute_reply.started":"2025-01-01T01:27:15.09867Z","shell.execute_reply":"2025-01-01T01:27:15.098688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nsub30 = pd.read_csv('/kaggle/input/v2-subs/submission_cbc_29.csv')\nsub31 = pd.read_csv('/kaggle/input/v2-subs/submission_cbc_33.csv')\nsub32 = pd.read_csv('/kaggle/input/v2-subs/submission_lgbc_29.csv')\nsub33 = pd.read_csv('/kaggle/input/v2-subs/submission_lgbc_34.csv')\nsub34 = pd.read_csv('/kaggle/input/v2-subs/submission_xgbc_23.csv')\nsub35 = pd.read_csv('/kaggle/input/v2-subs/submission_xgbc_27.csv')\nsub36 = pd.read_csv('/kaggle/input/v2-subs/submission_xgbc_32.csv')\nsub37 = pd.read_csv('/kaggle/working/submission_mix_ens2.csv') \n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.099982Z","iopub.status.idle":"2025-01-01T01:27:15.100326Z","shell.execute_reply.started":"2025-01-01T01:27:15.100173Z","shell.execute_reply":"2025-01-01T01:27:15.10019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nsub40 = pd.read_csv('/kaggle/working/submission_cbc_23.csv')\nsub41 = pd.read_csv('/kaggle/working/submission_lgbc_23.csv')\nsub42 = pd.read_csv('/kaggle/working/submission_xgbc_23.csv')\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.101583Z","iopub.status.idle":"2025-01-01T01:27:15.101894Z","shell.execute_reply.started":"2025-01-01T01:27:15.101746Z","shell.execute_reply":"2025-01-01T01:27:15.101761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n%%time\nfinal_pred_mix = np.array([])\nimport statistics\n#pred1=cbc_model.predict_proba(dropped_test_df)[:, 1]\n#pred2=lgbm_model.predict_proba(dropped_test_df)[:, 1]_bag_class\n#pred3=xgbc_model.predict_proba(dropped_test_df)[:, 1]\n\n#try concatenate instead? more efficient?\nfor i in range(0,len(test_df)):\n    final_pred_mix = np.append(final_pred_mix, statistics.mean([sub11[TARGET][i], sub10[TARGET][i],\n                                                        sub13[TARGET][i],\n                                                        sub14[TARGET][i],\n                                                       \n                                                       ]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.103423Z","iopub.status.idle":"2025-01-01T01:27:15.103742Z","shell.execute_reply.started":"2025-01-01T01:27:15.10358Z","shell.execute_reply":"2025-01-01T01:27:15.103595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#final_pred_mix = np.array([int(i) for i in final_pred_mix]) \n#note to self remember that I started this as a predict proba problem, so I have some past floats creept in\n#final_pred_mix","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.10474Z","iopub.status.idle":"2025-01-01T01:27:15.105094Z","shell.execute_reply.started":"2025-01-01T01:27:15.104905Z","shell.execute_reply":"2025-01-01T01:27:15.104922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#or:\n'''\nfinal_pred=(#(0.3*sub8['loan_status'])+\n            #0.1*sub16['loan_status'])+\n            (0.81*sub0[TARGET])+(0.14*sub1[TARGET])+(0.05*sub2[TARGET]))\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.106562Z","iopub.status.idle":"2025-01-01T01:27:15.107047Z","shell.execute_reply.started":"2025-01-01T01:27:15.106786Z","shell.execute_reply":"2025-01-01T01:27:15.106809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n%%time\n#averaging\n#NEEDS POST FIT MODEL\n'''\n\nfinal_pred_all=(sub0[TARGET]+sub1[TARGET]+sub2[TARGET]+\n           sub3[TARGET]+sub4[TARGET]+sub5[TARGET]+\n            sub6[TARGET]+sub7[TARGET]+sub8[TARGET]\n           )//9\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.108909Z","iopub.status.idle":"2025-01-01T01:27:15.109434Z","shell.execute_reply.started":"2025-01-01T01:27:15.109159Z","shell.execute_reply":"2025-01-01T01:27:15.109183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df40 = pd.DataFrame({\n        \"id\": test_df[ID],\n        TARGET: final_pred_mix\n        #TARGET: np.expm1(final_pred_mix)\n    })","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.111097Z","iopub.status.idle":"2025-01-01T01:27:15.111583Z","shell.execute_reply.started":"2025-01-01T01:27:15.111323Z","shell.execute_reply":"2025-01-01T01:27:15.111346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path=\"/kaggle/working/submission_mean_misc10_11_13_14.csv\"\n        \nsub_df40.to_csv(path, index=False)\nprint(f\"Submission exported to {path}\")\n!head '/kaggle/working/submission_mean_misc10_11_13_14.csv'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.11427Z","iopub.status.idle":"2025-01-01T01:27:15.114742Z","shell.execute_reply.started":"2025-01-01T01:27:15.114498Z","shell.execute_reply":"2025-01-01T01:27:15.114522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission exported to /kaggle/working/submission_mean_vr.csv\nid,Premium Amount\n1200000,884.7864997742638\n1200001,871.9303634804259\n1200002,824.531115396138\n1200003,744.7946546277635\n1200004,790.9906484125928\n1200005,816.2750612597836\n1200006,791.578075310682\n1200007,578.4422214410076\n1200008,198.66341854345825\nSubmission exported to /kaggle/working/submission_mean_cb3.csv\nid,Premium Amount\n1200000,891.8485449239532\n1200001,869.4793702635163\n1200002,821.3020384938616\n1200003,746.945061342927\n1200004,790.7119615465826\n1200005,818.0365750925454\n1200006,791.0174622379726\n1200007,579.5776853283035\n1200008,198.6094420302277\nSubmission exported to /kaggle/working/submission_mean_Z.csv\nid,Premium Amount\n1200000,883.3393718173851\n1200001,863.6008737853144\n1200002,825.9580493025622\n1200003,736.6474474867402\n1200004,791.746518279217\n1200005,816.991798140908\n1200006,809.8523099944649\n1200007,584.2609084625269\n1200008,199.3444878447822\nmean_misc3\nid,Premium Amount\n1200000,911.1934039435407\n1200001,930.6514628670174\n1200002,812.7505029825741\n1200003,796.3219603182035\n1200004,788.6027902195722\n1200005,902.0462821754811\n1200006,646.3984846555252\n1200007,508.7724652366173\n1200008,179.48334903913488\nSubmission exported to /kaggle/working/submission_mean_misc3vr3.csv\nid,Premium Amount\n1200000,854.9953108931406\n1200001,870.2826032503361\n1200002,809.3052643986986\n1200003,805.4469733095093\n1200004,784.3582322425227\n1200005,842.427556586926\n1200006,724.649600876882\n1200007,580.4320887585457\n1200008,184.29977942307335\nSubmission exported to /kaggle/working/submission_mean_misc11-14-3.csv\nid,Premium Amount\n1200000,924.1844316389466\n1200001,895.477670970379\n1200002,805.0655784516272\n1200003,768.2042074717439\n1200004,772.6043622305866\n1200005,864.9766550860783\n1200006,798.3069737263373\n1200007,589.6203704878357\n1200008,196.5003135790938\nSubmission exported to /kaggle/working/submission_mean_misc11thru14-3.csv\nid,Premium Amount\n1200000,897.442363174382\n1200001,899.5112950200878\n1200002,812.8201398487854\n1200003,782.4434278541345\n1200004,784.9769701183797\n1200005,865.3408564453104\n1200006,722.0111323861281\n1200007,554.0496322751069\n1200008,187.47434128057583\nSubmission exported to /kaggle/working/submission_mean_misc9thru14-3-0.csv\nid,Premium Amount\n1200000,894.8020083466417\n1200001,885.1257646007067\n1200002,816.535952055598\n1200003,766.0763377345479\n1200004,787.4110431039398\n1200005,847.8705272684451\n1200006,754.7413440869884\n1200007,565.8046598031253\n1200008,191.90540504969178\n\nSubmission exported to /kaggle/working/submission_mean_misc10thru14.csv\nid,Premium Amount\n1200000,907.5224598858123\n1200001,921.6042262961247\n1200002,826.5666051658172\n1200003,762.9226883529627\n1200004,798.5554766138131\n1200005,871.8911355026667\n1200006,671.0231236039209\n1200007,504.34696873385303\n1200008,185.8823460288314\nSubmission exported to /kaggle/working/submission_mean_misc11thru14plus3.csv\nid,Premium Amount\n1200000,897.442363174382\n1200001,899.5112950200878\n1200002,812.8201398487854\n1200003,782.4434278541345\n1200004,784.9769701183797\n1200005,865.3408564453104\n1200006,722.0111323861281\n1200007,554.0496322751069\n1200008,187.47434128057583\nSubmission exported to /kaggle/working/submission_mean_misc10_11_13_14.csv\nid,Premium Amount\n1200000,916.54597437764\n1200001,927.0922893562941\n1200002,821.2936872130716\n1200003,755.9612384803769\n1200004,799.7347825975291\n1200005,880.9664488158118\n1200006,692.4863325291345\n1200007,508.0329165655475\n1200008,192.64805096922294","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.116318Z","iopub.status.idle":"2025-01-01T01:27:15.116812Z","shell.execute_reply.started":"2025-01-01T01:27:15.11656Z","shell.execute_reply":"2025-01-01T01:27:15.116584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!head '/kaggle/input/first-runs/mode_4.csv'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.118083Z","iopub.status.idle":"2025-01-01T01:27:15.118562Z","shell.execute_reply.started":"2025-01-01T01:27:15.118306Z","shell.execute_reply":"2025-01-01T01:27:15.11833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!head '/kaggle/input/first-runs/mode_6.csv'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.11982Z","iopub.status.idle":"2025-01-01T01:27:15.120167Z","shell.execute_reply.started":"2025-01-01T01:27:15.119977Z","shell.execute_reply":"2025-01-01T01:27:15.119992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!head '/kaggle/input/first-runs/lgbm_3.csv'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.121244Z","iopub.status.idle":"2025-01-01T01:27:15.121608Z","shell.execute_reply.started":"2025-01-01T01:27:15.121405Z","shell.execute_reply":"2025-01-01T01:27:15.121421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bag_model_cbc = pickle.load(open('/kaggle/working/cbc0_23bag_model.pickle', 'rb'))\nbag_model_lgbc = pickle.load(open('/kaggle/working/lgbc0_23bag_model.pickle', 'rb'))\nbag_model_xgbc = pickle.load(open('/kaggle/working/xgbc0_23bag_model.pickle', 'rb'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.122685Z","iopub.status.idle":"2025-01-01T01:27:15.12299Z","shell.execute_reply.started":"2025-01-01T01:27:15.122842Z","shell.execute_reply":"2025-01-01T01:27:15.122856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n%%time\n#NEED PREFIT MODELS\n#voting classifier\nfrom sklearn.ensemble import VotingRegressor\n\n#vc_model_h = VotingClassifier(estimators=[('cbc', bag_model_cbc),('lgbm', bag_model_lgbc), ('xgbc', bag_model_xgbc)], voting='hard', verbose=True) #predict_proba is not available when voting='hard'\n#vc_model_h = VotingRegressor(estimators=[('cbc', cbc0),('lgbm', lgbc0), ('xgbc', xgbc0)], voting='hard', verbose=True) #predict_proba is not available when voting='hard'\n#vc_model_h = VotingRegressor(estimators=[('cbc', cbc0),('lgbm', lgbc0),('xgbc', xgbc0)], verbose=True) #predict_proba is not available when voting='hard'\n\nvc_model_h = VotingRegressor(estimators=[('cbc', cbc0),('lgbm', lgbc0),('xgbc', xgbc0)],weights=[0.8, 0.1, 0.1], verbose=True)\nvc_model_h.fit(X_train,y_train)\n#premember my y is log transformed\ny_pred = vc_model_h.predict(X_test)\nhard_vc_rmsle = rmsle_score(y_test, y_pred)\nprint(f\"Base RMSLE score: {np.round(hard_vc_rmsle,6)}\") \n\n#final_pred=vc_model.predict_proba(dropped_test_df)[:, 1]\nfinal_pred_h=vc_model_h.predict(dropped_test_df)\nwith open(pickle_path+'vc_model_cbc_lgbc_hard_1.pickle', 'wb') as to_write:\n    pickle.dump(vc_model_h, to_write)\n\n#Base RMSLE score: 1.050605\n#CPU times: user 30min 45s, sys: 16.7 s, total: 31min 1s\n#Wall time: 11min","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.124494Z","iopub.status.idle":"2025-01-01T01:27:15.124817Z","shell.execute_reply.started":"2025-01-01T01:27:15.124654Z","shell.execute_reply":"2025-01-01T01:27:15.124669Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Base RMSLE score: 1.051001\nBase RMSLE score: 1.049963 w weights","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.125821Z","iopub.status.idle":"2025-01-01T01:27:15.126176Z","shell.execute_reply.started":"2025-01-01T01:27:15.125982Z","shell.execute_reply":"2025-01-01T01:27:15.125997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df_vc2h = pd.DataFrame({\n        \"id\": test_df[\"id\"],\n        TARGET: np.expm1(final_pred_h)\n    })","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.127607Z","iopub.status.idle":"2025-01-01T01:27:15.127923Z","shell.execute_reply.started":"2025-01-01T01:27:15.127773Z","shell.execute_reply":"2025-01-01T01:27:15.127789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path=\"/kaggle/working/submission_vr_3.csv\"\n        \nsub_df_vc2h.to_csv(path, index=False)\nprint(f\"Submission exported to {path}\")\n!head /kaggle/working/submission_vr_3.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.128828Z","iopub.status.idle":"2025-01-01T01:27:15.129185Z","shell.execute_reply.started":"2025-01-01T01:27:15.128989Z","shell.execute_reply":"2025-01-01T01:27:15.129005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef use_bagging_classifier(the_classifier, num_of_estimators, classifier_name):\n    this_bag_model = BaggingClassifier(base_estimator=the_classifier, n_estimators=num_of_estimators, random_state=42, verbose=1)\n    this_bag_model.fit(X_train, y_train)\n    final_bag_pred = bag_model_cbc.predict(dropped_test_df) #remeber change for  predict_proba etc..\n    with open(pickle_path+classifier_name+'_bag_model.pickle', 'wb') as to_write:\n    pickle.dump(this_bag_model, to_write)\n    return final_bag_pred\n\ndef create_sub_df(final_preds):\n    submit = pd.DataFrame({\n        \"id\": test_df[\"id\"],\n        TARGET: final_preds\n    })\n    return submit\n\ndef write_pres_file(file,path):\n    file.to_csv(path, index=False)\n    print(f\"Submission exported to {path}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.130323Z","iopub.status.idle":"2025-01-01T01:27:15.130698Z","shell.execute_reply.started":"2025-01-01T01:27:15.13049Z","shell.execute_reply":"2025-01-01T01:27:15.130507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nbag_preds = use_bagging_classifier(cbco, 50, 'cbc_0')\nthe_submit = create_sub_df(bag_preds)\npath=\"/kaggle/working/submission_bag_cbc_0.csv\"\nwrite_pres_file(the_submit,path)\n!head path\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.132594Z","iopub.status.idle":"2025-01-01T01:27:15.133092Z","shell.execute_reply.started":"2025-01-01T01:27:15.132821Z","shell.execute_reply":"2025-01-01T01:27:15.132846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#cbc0 = pickle.load(open('/kaggle/working/cbc_clf0.pickle', 'rb'))\n#lgbc0 = pickle.load(open('/kaggle/working/lgbc_clf0.pickle', 'rb'))\n#xgbc0 = pickle.load(open('/kaggle/working/xgbc_clf0.pickle', 'rb'))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.134144Z","iopub.status.idle":"2025-01-01T01:27:15.13462Z","shell.execute_reply.started":"2025-01-01T01:27:15.134366Z","shell.execute_reply":"2025-01-01T01:27:15.1344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n%%time\n#NEED PREFIT MODELS\nfrom sklearn.ensemble import BaggingRegressor\n                \nbag_model_cbc = BaggingRegressor(estimator=xgbc0, n_estimators=40, random_state=42, verbose=1)#, oob_score=True)\n\nbag_model_cbc.fit(X_train, y_train_trans)\ny_pred_trans = bag_model_cbc.predict(X_test)\nbag_cbc_rmsle = rmsle_score(y_test_trans, y_pred_trans)\nprint(f\"Base RMSLE score: {np.round(bag_cbc_rmsle,6)}\") \n\nwith open(pickle_path+'cbc0_bag1_model.pickle', 'wb') as to_write:\n    pickle.dump(bag_model_cbc, to_write)\n    \n\nfinal_pred_bag_cbc = bag_model_cbc.predict(dropped_test_df)#[:, 1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.135853Z","iopub.status.idle":"2025-01-01T01:27:15.136339Z","shell.execute_reply.started":"2025-01-01T01:27:15.136096Z","shell.execute_reply":"2025-01-01T01:27:15.13612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df3 = pd.DataFrame({\n        \"id\": test_df[\"id\"],\n        TARGET: np.expm1(final_pred_bag_cbc)\n    })","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.137592Z","iopub.status.idle":"2025-01-01T01:27:15.138083Z","shell.execute_reply.started":"2025-01-01T01:27:15.137812Z","shell.execute_reply":"2025-01-01T01:27:15.137836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path=\"/kaggle/working/submission_cbc_bag1.csv\"\n        \nsub_df3.to_csv(path, index=False)\nprint(f\"Submission exported to {path}\")\n!head /kaggle/working/submission_cbc_bag1.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.139803Z","iopub.status.idle":"2025-01-01T01:27:15.140295Z","shell.execute_reply.started":"2025-01-01T01:27:15.140048Z","shell.execute_reply":"2025-01-01T01:27:15.140073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n#NEED PREFIT MODELS\n\n                \nbag_model_lgbc = BaggingClassifier(base_estimator=lgbc0, n_estimators=60, random_state=42, verbose=1)#, oob_score=True)\n\nbag_model_lgbc.fit(X_train, y_train)\nfinal_pred_bag_lgbc = bag_model_lgbc.predict(dropped_test_df)#[:, 1]\n\nwith open(pickle_path+'lgbc0_23bag_model.pickle', 'wb') as to_write:\n    pickle.dump(bag_model_lgbc, to_write)\n    \ntest_acc = accuracy_score(y_test,bag_model_lgbc.predict(X_test))\nprint(test_acc)\n\n#0.9388770433546553\n#0.9389907604832978\n#0.9389054726368159","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.141725Z","iopub.status.idle":"2025-01-01T01:27:15.142216Z","shell.execute_reply.started":"2025-01-01T01:27:15.141947Z","shell.execute_reply":"2025-01-01T01:27:15.141971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df4 = pd.DataFrame({\n        \"id\": test_df[\"id\"],\n        TARGET: final_pred_bag_lgbc\n    })","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.143669Z","iopub.status.idle":"2025-01-01T01:27:15.144151Z","shell.execute_reply.started":"2025-01-01T01:27:15.143889Z","shell.execute_reply":"2025-01-01T01:27:15.143912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path=\"/kaggle/working/submission_23lgbc_bag.csv\"\n        \nsub_df4.to_csv(path, index=False)\nprint(f\"Submission exported to {path}\")\n!head /kaggle/working/submission_23lgbc_bag.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.145341Z","iopub.status.idle":"2025-01-01T01:27:15.145807Z","shell.execute_reply.started":"2025-01-01T01:27:15.145569Z","shell.execute_reply":"2025-01-01T01:27:15.145593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n#NEED PREFIT MODELS\n\n                \nbag_model_xgbc = BaggingClassifier(base_estimator=xgbc0, n_estimators=60, random_state=42, verbose=1)#, oob_score=True)\n\nbag_model_xgbc.fit(X_train, y_train)\nfinal_pred_bag_xgbc = bag_model_xgbc.predict(dropped_test_df)#[:, 1]\n\nwith open(pickle_path+'xgbc0_23bag_model.pickle', 'wb') as to_write:\n    pickle.dump(bag_model_xgbc, to_write)\n    \ntest_acc = accuracy_score(y_test,bag_model_xgbc.predict(X_test))\nprint(test_acc)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.147272Z","iopub.status.idle":"2025-01-01T01:27:15.147744Z","shell.execute_reply.started":"2025-01-01T01:27:15.147504Z","shell.execute_reply":"2025-01-01T01:27:15.147526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df5 = pd.DataFrame({\n        \"id\": test_df[\"id\"],\n        TARGET: final_pred_bag_xgbc\n    })","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.149104Z","iopub.status.idle":"2025-01-01T01:27:15.149441Z","shell.execute_reply.started":"2025-01-01T01:27:15.149273Z","shell.execute_reply":"2025-01-01T01:27:15.14929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path=\"/kaggle/working/submission_23xgbc_bag.csv\"\n        \nsub_df5.to_csv(path, index=False)\nprint(f\"Submission exported to {path}\")\n!head /kaggle/working/submission_23xgbc_bag.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.150306Z","iopub.status.idle":"2025-01-01T01:27:15.150636Z","shell.execute_reply.started":"2025-01-01T01:27:15.15048Z","shell.execute_reply":"2025-01-01T01:27:15.150496Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n%%time\n#NEED PREFIT MODELS\n\nfrom sklearn.ensemble import AdaBoostClassifier\nada_model_cbc = AdaBoostClassifier(estimator=bag_model_cbc, n_estimators=60,learning_rate=0.01, random_state=42,) #def n_est=50\nada_model_cbc.fit(X_train, y_train)\n#ada_model.score(X_test,y_test)\n#print(\"The accuracy of the model on validation set is\", ada_model.score(X_test,y_test))\n#final_pred = ada_model.predict_proba(dropped_test_df)[:, 1]\nfinal_pred_ada_cbc = ada_model_cbc.predict(dropped_test_df)#[:, 1]\n\ntest_acc = accuracy_score(y_test,ada_model_cbc.predict(X_test))\nprint(test_acc)\nwith open(pickle_path+'cbc_bag_ada_model.pickle', 'wb') as to_write:\n    pickle.dump(ada_model_cbc, to_write)\n\n'''\nParameters\n\nbase_estimators: It helps to specify the type of base estimator, that is, the machine learning algorithm to be used as base learner.\n\nn_estimators: It defines the number of base estimators. The default value is 10, but you should keep a higher value to get better performance.\n\nlearning_rate: This parameter controls the contribution of the estimators in the final combination. There is a trade-off between learning_rate and n_estimators.\n\nmax_depth: Defines the maximum depth of the individual estimator. Tune this parameter for best performance.\n\nn_jobs: Specifies the number of processors it is allowed to use. Set value to -1 for maximum processors allowed.\n\nrandom_state : An integer value to specify the random data split. A definite value of random_state will always produce same results if given with same parameters and training data.\n'''\n#0.923724235963042","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.15192Z","iopub.status.idle":"2025-01-01T01:27:15.152298Z","shell.execute_reply.started":"2025-01-01T01:27:15.152099Z","shell.execute_reply":"2025-01-01T01:27:15.152116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df6 = pd.DataFrame({\n        \"id\": test_df[\"id\"],\n        TARGET: final_pred_ada_cbc\n    })","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.153754Z","iopub.status.idle":"2025-01-01T01:27:15.154107Z","shell.execute_reply.started":"2025-01-01T01:27:15.153919Z","shell.execute_reply":"2025-01-01T01:27:15.153935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path=\"/kaggle/working/submission_2cbc_ada_bag.csv\"\n        \nsub_df6.to_csv(path, index=False)\nprint(f\"Submission exported to {path}\")\n!head /kaggle/working/submission_2cbc_ada_bag.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.155572Z","iopub.status.idle":"2025-01-01T01:27:15.155956Z","shell.execute_reply.started":"2025-01-01T01:27:15.15575Z","shell.execute_reply":"2025-01-01T01:27:15.155767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n#NEED PREFIT MODELS\n\nada_model_lgbc = AdaBoostClassifier(estimator=bag_model_lgbc, n_estimators=60,learning_rate=0.01, random_state=42,) #def n_est=50\nada_model_lgbc.fit(X_train, y_train)\n#ada_model.score(X_test,y_test)\n#print(\"The accuracy of the model on validation set is\", ada_model.score(X_test,y_test))\n#final_pred = ada_model.predict_proba(dropped_test_df)[:, 1]\nfinal_pred_ada_lgbc = ada_model_lgbc.predict(dropped_test_df)#[:, 1]\n\ntest_acc = accuracy_score(y_test,ada_model_lgbc.predict(X_test))\nprint(test_acc)\nwith open(pickle_path+'lgbc_bag_ada_model.pickle', 'wb') as to_write:\n    pickle.dump(ada_model_lgbc, to_write)\n#0.9215067519545131","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.157753Z","iopub.status.idle":"2025-01-01T01:27:15.15811Z","shell.execute_reply.started":"2025-01-01T01:27:15.157922Z","shell.execute_reply":"2025-01-01T01:27:15.157938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df7 = pd.DataFrame({\n        \"id\": test_df[\"id\"],\n        TARGET: final_pred_ada_lgbc\n    })","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.158905Z","iopub.status.idle":"2025-01-01T01:27:15.159254Z","shell.execute_reply.started":"2025-01-01T01:27:15.15909Z","shell.execute_reply":"2025-01-01T01:27:15.159108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path=\"/kaggle/working/submission_2lgbc_ada_bag.csv\"\n        \nsub_df7.to_csv(path, index=False)\nprint(f\"Submission exported to {path}\")\n!head /kaggle/working/submission_2lgbc_ada_bag.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.161373Z","iopub.status.idle":"2025-01-01T01:27:15.161701Z","shell.execute_reply.started":"2025-01-01T01:27:15.161539Z","shell.execute_reply":"2025-01-01T01:27:15.161555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n#NEED PREFIT MODELS\n\nada_model_xgbc = AdaBoostClassifier(estimator=bag_model_xgbc, n_estimators=60,learning_rate=0.01, random_state=42,) #def n_est=50\nada_model_xgbc.fit(X_train, y_train)\n#ada_model.score(X_test,y_test)\n#print(\"The accuracy of the model on validation set is\", ada_model.score(X_test,y_test))\n#final_pred = ada_model.predict_proba(dropped_test_df)[:, 1]\nfinal_pred_ada_xgbc = ada_model_xgbc.predict(dropped_test_df)#[:, 1]\n\ntest_acc = accuracy_score(y_test,ada_model_xgbc.predict(X_test))\nprint(test_acc)\nwith open(pickle_path+'xgbc_bag_ada_model.pickle', 'wb') as to_write:\n    pickle.dump(ada_model_xgbc, to_write)\n#0.8182800284292822","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.162782Z","iopub.status.idle":"2025-01-01T01:27:15.1632Z","shell.execute_reply.started":"2025-01-01T01:27:15.163002Z","shell.execute_reply":"2025-01-01T01:27:15.163019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df8 = pd.DataFrame({\n        \"id\": test_df[\"id\"],\n        TARGET: final_pred_ada_xgbc\n    })","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.164264Z","iopub.status.idle":"2025-01-01T01:27:15.164615Z","shell.execute_reply.started":"2025-01-01T01:27:15.164437Z","shell.execute_reply":"2025-01-01T01:27:15.164454Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path=\"/kaggle/working/submission_2xgbc_ada_bag.csv\"\n        \nsub_df8.to_csv(path, index=False)\nprint(f\"Submission exported to {path}\")\n!head /kaggle/working/submission_2xgbc_ada_bag.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.165779Z","iopub.status.idle":"2025-01-01T01:27:15.166111Z","shell.execute_reply.started":"2025-01-01T01:27:15.165931Z","shell.execute_reply":"2025-01-01T01:27:15.165946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef prediction_to_kaggle_format(model, threshold=0.5):\n\n    #prob_pred = model.predict(dropped_test_df)#[:,0]\n\n    #prob_pred = model.predict_proba(dropped_test_df)[:, 1]\n    prob_pred = model.predict(dropped_test_df)#[:, 1]\n    ####################################\n    #pred_original = np.expm1(prob_pred) \n    pred_original = prob_pred \n    ####################################\n    #print(prob_pred)\n    #prob_pred = (np.around(model.predict_proba(the_test_df)[:, 1], 1))\n\n    return pd.DataFrame({\n        \"id\": test_df[\"id\"],\n        TARGET: np.expm1(pred_original)#(proba_survive >= threshold).astype(int)\n    })\n\ndef make_submission_gbc(kaggle_predictions, model='none'):\n    if model=='lgbc':\n        path=\"/kaggle/working/submission_lgbc_3.csv\"\n    elif model=='cbc':\n        path=\"/kaggle/working/submission_cbc_3.csv\"\n    elif model=='xgbc':\n        path=\"/kaggle/working/submission_xgbc_3.csv\"\n    else:\n        path=\"/kaggle/working/submission_generic.csv\"\n        \n    kaggle_predictions.to_csv(path, index=False)\n    print(f\"Submission exported to {path}\")\n\n'''\ndef make_submission_cbc(kaggle_predictions):\n    path=\"/kaggle/working/submission_grid.csv\"\n    kaggle_predictions.to_csv(path, index=False)\n    print(f\"Submission exported to {path}\")\n'''\n\nkaggle_predictions_lgbc0 = prediction_to_kaggle_format(lgbc0)\nkaggle_predictions_cbc0 = prediction_to_kaggle_format(cbc0)\nkaggle_predictions_xgbc0 = prediction_to_kaggle_format(xgbc0)\n\n#kaggle_predictions_lgbc0['class'] = kaggle_predictions_lgbc0['class'].map(rev_target_map)\n#kaggle_predictions_xgbc0['class'] = kaggle_predictions_xgbc0['class'].map(rev_target_map)\n#kaggle_predictions_cbc0['class'] = kaggle_predictions_cbc0['class'].map(rev_target_map)\n#TARGET = 'Depression'\n\nmake_submission_gbc(kaggle_predictions_lgbc0, 'lgbc')\nmake_submission_gbc(kaggle_predictions_cbc0, 'cbc')\nmake_submission_gbc(kaggle_predictions_xgbc0, 'xgbc')\n#output.head()\n#make_submission(output)\n!head /kaggle/working/submission_cbc_3.csv\n!head /kaggle/working/submission_lgbc_3.csv\n!head /kaggle/working/submission_xgbc_3.csv\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.167863Z","iopub.status.idle":"2025-01-01T01:27:15.168355Z","shell.execute_reply.started":"2025-01-01T01:27:15.168108Z","shell.execute_reply":"2025-01-01T01:27:15.168133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#or do I :\n'''\nrev_target_map = {0:'e',1:'p',\n             }\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.169762Z","iopub.status.idle":"2025-01-01T01:27:15.170256Z","shell.execute_reply.started":"2025-01-01T01:27:15.169985Z","shell.execute_reply":"2025-01-01T01:27:15.170008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\ndef prediction_to_kaggle_format(model, threshold=0.5):\n\n    prob_pred = model.predict(dropped_test_df)#[:,0]\n\n    #prob_pred = model.predict_proba(dropped_test_df)[:, 1]\n    #print(prob_pred)\n    #prob_pred = (np.around(model.predict_proba(the_test_df)[:, 1], 1))\n\n    return pd.DataFrame({\n        \"id\": test_df[\"id\"],\n        \"class\": prob_pred#(proba_survive >= threshold).astype(int)\n    })\n\n\n\ndef make_submission_base(kaggle_predictions):\n    path=\"/kaggle/working/submission_lgbc0.csv\"\n    kaggle_predictions.to_csv(path, index=False)\n    print(f\"Submission exported to {path}\")\n\n\nkaggle_predictions_base = prediction_to_kaggle_format(lgbc0)\n\nkaggle_predictions_base['class'] = kaggle_predictions_base['class'].map(rev_target_map)\n\n\nmake_submission_base(kaggle_predictions_base)\n#output.head()\n#make_submission(output)\n!head /kaggle/working/submission_lgbc0.csv\n\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T01:27:15.171937Z","iopub.status.idle":"2025-01-01T01:27:15.172447Z","shell.execute_reply.started":"2025-01-01T01:27:15.17219Z","shell.execute_reply":"2025-01-01T01:27:15.172214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}