{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd \n\npath = \"playground-series-s4e12/train.csv\"\ndf = pd.read_csv(path, index_col =0)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.dropna()\n# We re-encode some variables and drop some other\ndf[\"Gender\"] = df[\"Gender\"].map(lambda a : 0 if a == \"Female\" else 1)\n\nm_status = {\"Single\":0,\"Married\":1,\"Divorced\":2}\ndf[\"Marital Status\"] = df[\"Marital Status\"].map(lambda status : m_status[status])\n\ned_lev = {\"High School\": 0, \"Bachelor's\":1,\n         \"Master's\":2, \"PhD\": 3}\ndf[\"Education Level\"] = df[\"Education Level\"].map(lambda ed : ed_lev[ed])\n\nocc_dict = {\"Unemployed\": 0, \"Employed\":2,\n         \"Self-Employed\":1}\ndf[\"Occupation\"] = df[\"Occupation\"].map(lambda occ : occ_dict[occ])\n\nloc_dict = {\"Rural\": 0, \"Suburban\":1,\n         \"Urban\":2}\ndf[\"Location\"] = df[\"Location\"].map(lambda loc : loc_dict[loc])\n\nloc_dict = {\"Basic\": 0, \"Comprehensive\":1,\n         \"Premium\":2}\ndf[\"Policy Type\"] = df[\"Policy Type\"].map(lambda loc : loc_dict[loc])\n\nloc_dict = {\"Poor\": 0, \"Average\":1,\n         \"Good\":2}\ndf[\"Customer Feedback\"] = df[\"Customer Feedback\"].map(lambda loc : loc_dict[loc])\n\nloc_dict = {\"No\": 0, \"Yes\":1}\ndf[\"Smoking Status\"] = df[\"Smoking Status\"].map(lambda loc : loc_dict[loc])\n\nloc_dict = {\"Rarely\": 0, \"Monthly\":1,\"Weekly\":2,\"Daily\":3}\ndf[\"Exercise Frequency\"] = df[\"Exercise Frequency\"].map(lambda loc : loc_dict[loc])\n\nloc_dict = {\"Condo\": 0, \"Apartment\":1,\"House\":2}\ndf[\"Property Type\"] = df[\"Property Type\"].map(lambda loc : loc_dict[loc])\n\ndf[\"Policy Start Date\"] = pd.to_datetime(df[\"Policy Start Date\"]).dt.year","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sympy import symbols, Add, Mul, sin, cos, Pow\nimport random\n\n# Define possible operations\noperations = [Add, Mul, sin, cos, Pow]\n\n# A Recursive function that'll generate as many equation as we need\ndef generate_random_expression(trial,variables, operations, node_name):\n    #### ---- Here is generated the terminal node of the recursion ---- ####\n    # This will allow us to break the recursion\n    if trial.suggest_int(node_name+\"_stop\",0,1):\n        # A terminal node will include a coefficient and a variable\n        coef = trial.suggest_float(node_name+\"_coef\", -10, 10)\n        var_choosen =  trial.suggest_categorical(node_name+\"_variable\",\n                                          list(range(len(variables))))\n        # Send a multiplication between the coefficient and the chosen variable\n        return Mul(coef,variables[var_choosen])\n\n    #### ---- Here are generated intermediate nodes of the recursion ---- ####\n    \n    # We select an operator Add, Mul, sin, cos OR Pow\n    operation = trial.suggest_categorical(node_name+\"_op\",\n                                        list(range(len(operations))))\n    operation = operations[operation]\n    \n    if operation in [Add, Mul]:\n        # The addition and Multiplication need us to get 2 variable\n        # We continue going into the depth of the recursion to find them !\n        return operation(generate_random_expression(trial, \n                                variables, operations, node_name+\"_op1\"),\n                         generate_random_expression(trial, \n                                variables, operations, node_name+\"_op2\"))\n    elif operation == Pow:\n        # For the Power we break the equation too, but no need to do so\n        coef = trial.suggest_int(node_name+\"_pow_coef\", 0, 4)\n        var_choosen =  trial.suggest_categorical(node_name+\"_pow_variable\",\n                                        list(range(len(variables))))\n        return Pow(variables[var_choosen],coef)\n    else:\n        # Here we compute more depth for the sine or cosine\n        return operation(generate_random_expression(trial,variables, operations, node_name+\"_op\"))","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nfrom sympy import lambdify, symbols\nimport numpy as np \ndef objectif(trial, X_train, y_train):\n    # Here I only have 8 features so I generate 8 variable using SymPy's symbols function\n    all_sym = [x for x in symbols(\" \".join([f'x{i}' for i in range(X_train.shape[1])]))]\n    \n    # Here we generate our expression using the previous function\n    expr = generate_random_expression(trial,\n                all_sym, operations,\"node_0\")\n    # We add an intercept\n    intercept = trial.suggest_float(\"intercept\",-10,10)\n    expr = Add(expr,intercept)\n    func = lambdify(all_sym, expr, modules='numpy')\n    y_pred = np.array([func(*x_train) for x_train in X_train])\n    score = mean_squared_error(y_train, y_pred)\n    return score","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test=  train_test_split(df.iloc[:,:-1],df.iloc[:,-1], shuffle=True, train_size=0.7)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import optuna\nfrom optuna.samplers import QMCSampler\nstudy_qmc = optuna.create_study(direction=\"minimize\", \n                sampler=QMCSampler())\nstudy_qmc.optimize(lambda trial : objectif(trial, X_train.to_numpy(),y_train.to_numpy()), \n                n_trials=1_000, n_jobs=4)","metadata":{"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"study_qmc.best_params","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reconstruct_equation(operations, variables, parametres, node_name):\n    #print(parametres, node_name)\n    find_variable = lambda list_in, var_name : [v for (k,v) in list_in if k.startswith(var_name)][0]\n    try:\n        #print([(k.split(node_name),v) for (k,v) in parametres.items() if k.startswith(node_name)])\n        parametres = [(k[len(node_name):],v) for (k,v) in parametres.items() if k.startswith(node_name)]\n        #print(parametres, node_name)\n    except:\n        parametres = [(k[len(node_name):],v) for (k,v) in parametres.items() if k.startswith(node_name)]\n        print(\"err_1\",parametres, node_name)\n\n    try:\n        stop = find_variable(parametres,\"_stop\")\n    except:\n        stop = find_variable(parametres,\"_stop\")\n        print(\"error\",parametres)\n\n    if stop:\n        coef = find_variable(parametres,\"_coef\")\n        var_choosen = find_variable(parametres, \"_variable\")\n        return Mul(coef, variables[var_choosen])\n\n    operation = find_variable(parametres, \"_op\")\n    operation = operations[operation]\n    if operation in [Add, Mul]:\n        return operation(reconstruct_equation(operations, variables, {k : v for (k,v) in parametres if k.startswith(\"_op1\")}, \"_op1\"),\n                         reconstruct_equation(operations, variables, {k : v for (k,v) in parametres if k.startswith(\"_op2\")}, \"_op2\"))\n    elif operation == Pow:\n        coef = find_variable(parametres,\"_pow_coef\")\n        var_choosen =  find_variable(parametres,\"_pow_variable\")\n        return Pow(variables[var_choosen],coef)\n    else:\n        return operation(reconstruct_equation(operations, variables, {k : v for (k,v) in parametres if ((k.startswith(\"_op\")) and (not k.startswith(\"_op1\")) and(not k.startswith(\"_op2\")) and (not k == \"_op\"))}, \"_op\"))\n\nreconstruct_equation(operations, [x for x in symbols(\" \".join([f'x{i}' for i in range(X_train.shape[1])]))],study_qmc.best_params, \"node_0\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.columns[[12,3]]","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"6.4*np.sin(df[\"Marital Status\"]).unique()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Here the single and divorced are considered as similar ! ","metadata":{}},{"cell_type":"code","source":"mean_squared_error(y_test, len(y_test)*[np.mean(y_train)])","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\n\nLR = LinearRegression()\n\nLR.fit(X_train,y_train)\nmean_squared_error(y_test, LR.predict(X_test))","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The model is inefficient but fun ride anyway","metadata":{}}]}