{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:38:38.227568Z","iopub.execute_input":"2024-12-10T18:38:38.22798Z","iopub.status.idle":"2024-12-10T18:38:38.23567Z","shell.execute_reply.started":"2024-12-10T18:38:38.227944Z","shell.execute_reply":"2024-12-10T18:38:38.234622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:38:38.238137Z","iopub.execute_input":"2024-12-10T18:38:38.238604Z","iopub.status.idle":"2024-12-10T18:38:46.366723Z","shell.execute_reply.started":"2024-12-10T18:38:38.238554Z","shell.execute_reply":"2024-12-10T18:38:46.365616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df_train.shape,df_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:38:46.36864Z","iopub.execute_input":"2024-12-10T18:38:46.368989Z","iopub.status.idle":"2024-12-10T18:38:46.374506Z","shell.execute_reply.started":"2024-12-10T18:38:46.368957Z","shell.execute_reply":"2024-12-10T18:38:46.373343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:38:46.375889Z","iopub.execute_input":"2024-12-10T18:38:46.376493Z","iopub.status.idle":"2024-12-10T18:38:46.407449Z","shell.execute_reply.started":"2024-12-10T18:38:46.376427Z","shell.execute_reply":"2024-12-10T18:38:46.40628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_col = 'Premium Amount'\nid_col = 'id'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:38:46.410163Z","iopub.execute_input":"2024-12-10T18:38:46.410605Z","iopub.status.idle":"2024-12-10T18:38:46.420275Z","shell.execute_reply.started":"2024-12-10T18:38:46.410558Z","shell.execute_reply":"2024-12-10T18:38:46.419126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_cols = list(df_train.select_dtypes(include=['object', 'category']).columns)\nnum_cols = [col for col in df_train.columns if col not in (cat_cols + [target_col] + [id_col])]\nprint(f\"Categorical columns = {cat_cols} \\n\")\nprint(f\"Numerical columns = {num_cols}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:38:46.421459Z","iopub.execute_input":"2024-12-10T18:38:46.421836Z","iopub.status.idle":"2024-12-10T18:38:46.589432Z","shell.execute_reply.started":"2024-12-10T18:38:46.421801Z","shell.execute_reply":"2024-12-10T18:38:46.588153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_null_cols = [col for col in cat_cols if df_train[col].isnull().sum() > 0]\nnum_null_cols = [col for col in num_cols if df_train[col].isnull().sum() > 0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:38:46.591015Z","iopub.execute_input":"2024-12-10T18:38:46.591498Z","iopub.status.idle":"2024-12-10T18:38:47.206364Z","shell.execute_reply.started":"2024-12-10T18:38:46.591448Z","shell.execute_reply":"2024-12-10T18:38:47.205278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_null_cols, num_null_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:38:47.207617Z","iopub.execute_input":"2024-12-10T18:38:47.207918Z","iopub.status.idle":"2024-12-10T18:38:47.216204Z","shell.execute_reply.started":"2024-12-10T18:38:47.207888Z","shell.execute_reply":"2024-12-10T18:38:47.21514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import warnings\nimport matplotlib.pyplot as plt\ndef plot_num(df,col):\n    with warnings.catch_warnings():\n        warnings.simplefilter(\"ignore\", FutureWarning)\n        \n        data_column = df[col]\n        data_df = pd.DataFrame({col: data_column})\n\n        plt.figure(figsize=(8, 6))\n        sns.histplot(data=data_df, x=col, multiple=\"stack\")\n        plt.xlabel(col)\n        plt.ylabel('Density')\n        plt.title(f'Distribution {col}')\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:38:47.218289Z","iopub.execute_input":"2024-12-10T18:38:47.218803Z","iopub.status.idle":"2024-12-10T18:38:47.234379Z","shell.execute_reply.started":"2024-12-10T18:38:47.218725Z","shell.execute_reply":"2024-12-10T18:38:47.233142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import warnings\ndef plot_cat(df,col):\n    with warnings.catch_warnings():\n        warnings.simplefilter(\"ignore\", FutureWarning)\n        \n        data_column = df[col]\n        data_target = df[target_col]\n        data_df = pd.DataFrame({col: data_target, 'target': data_column})\n\n        plt.figure(figsize=(8, 6))\n        sns.histplot(data=data_df, x=col, hue='target', kde=True, multiple=\"stack\")\n        plt.xlabel(col)\n        plt.ylabel('Density')\n        plt.title(f'{col} Distribution by {target_col}')\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:38:47.235937Z","iopub.execute_input":"2024-12-10T18:38:47.236402Z","iopub.status.idle":"2024-12-10T18:38:47.252636Z","shell.execute_reply.started":"2024-12-10T18:38:47.236354Z","shell.execute_reply":"2024-12-10T18:38:47.251498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import seaborn as sns\n\n# for col in cat_cols:\n#     plot_cat(df_train,col)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:38:47.256078Z","iopub.execute_input":"2024-12-10T18:38:47.256469Z","iopub.status.idle":"2024-12-10T18:38:47.26441Z","shell.execute_reply.started":"2024-12-10T18:38:47.256436Z","shell.execute_reply":"2024-12-10T18:38:47.263215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.describe(include=\"all\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:38:47.265916Z","iopub.execute_input":"2024-12-10T18:38:47.266387Z","iopub.status.idle":"2024-12-10T18:38:50.055121Z","shell.execute_reply.started":"2024-12-10T18:38:47.26634Z","shell.execute_reply":"2024-12-10T18:38:50.054135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:38:50.05663Z","iopub.execute_input":"2024-12-10T18:38:50.057087Z","iopub.status.idle":"2024-12-10T18:38:50.065022Z","shell.execute_reply.started":"2024-12-10T18:38:50.057018Z","shell.execute_reply":"2024-12-10T18:38:50.063854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def describe_col(df, col):\n    value_counts = df[col].value_counts()\n    null_values = df[col].isnull().sum()\n    unique_values = df[col].nunique()\n    print(\"=\"*20)\n    print(f\"Describtion of {col}\")\n    print(\"Value counts: \")\n    print(value_counts)\n    print(f\"Null values: {null_values}\")\n    print(f\"Unique values: {unique_values}\")\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:38:50.0666Z","iopub.execute_input":"2024-12-10T18:38:50.06708Z","iopub.status.idle":"2024-12-10T18:38:50.076214Z","shell.execute_reply.started":"2024-12-10T18:38:50.067011Z","shell.execute_reply":"2024-12-10T18:38:50.075242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in df_train.columns:\n   describe_col(df_train,col)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:38:50.077458Z","iopub.execute_input":"2024-12-10T18:38:50.077798Z","iopub.status.idle":"2024-12-10T18:38:53.729354Z","shell.execute_reply.started":"2024-12-10T18:38:50.077766Z","shell.execute_reply":"2024-12-10T18:38:53.728281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.drop(columns=['id'],inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:38:53.730766Z","iopub.execute_input":"2024-12-10T18:38:53.731206Z","iopub.status.idle":"2024-12-10T18:38:53.99061Z","shell.execute_reply.started":"2024-12-10T18:38:53.731157Z","shell.execute_reply":"2024-12-10T18:38:53.989588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndef apply_feature_changes(df):\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    \n    df['Policy Start Date_year'] = df['Policy Start Date'].dt.year\n    df['Policy Start Date_month'] = df['Policy Start Date'].dt.month\n    df['Policy Start Date_day_of_month'] = df['Policy Start Date'].dt.day\n    \n    df['Policy Start Date_day'] = (df['Policy Start Date'] - pd.Timestamp('1970-01-01')).dt.days\n\n    df = df.drop(columns=['Policy Start Date'])\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:39:36.997857Z","iopub.execute_input":"2024-12-10T18:39:36.99874Z","iopub.status.idle":"2024-12-10T18:39:37.006387Z","shell.execute_reply.started":"2024-12-10T18:39:36.998685Z","shell.execute_reply":"2024-12-10T18:39:37.005256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = apply_feature_changes(df_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:39:37.153817Z","iopub.execute_input":"2024-12-10T18:39:37.154231Z","iopub.status.idle":"2024-12-10T18:39:37.612603Z","shell.execute_reply.started":"2024-12-10T18:39:37.154195Z","shell.execute_reply":"2024-12-10T18:39:37.611369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_cols = list(df_train.select_dtypes(include=['object', 'category']).columns)\nnum_cols = [col for col in df_train.columns if col not in (cat_cols) and  col != 'Premium Amount']\nprint(f\"Categorical columns = {cat_cols} \\n\")\nprint(f\"Numerical columns = {num_cols}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:39:37.614581Z","iopub.execute_input":"2024-12-10T18:39:37.614936Z","iopub.status.idle":"2024-12-10T18:39:38.184307Z","shell.execute_reply.started":"2024-12-10T18:39:37.614902Z","shell.execute_reply":"2024-12-10T18:39:38.183028Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ordinal_cols = ['Customer Feedback','Education Level','Policy Type','Location','Customer Feedback','Exercise Frequency']\n\none_hot_cols = [col for col in cat_cols if col not in ordinal_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:39:38.185594Z","iopub.execute_input":"2024-12-10T18:39:38.185975Z","iopub.status.idle":"2024-12-10T18:39:38.191527Z","shell.execute_reply.started":"2024-12-10T18:39:38.185942Z","shell.execute_reply":"2024-12-10T18:39:38.190252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ordinal_categories = {\n    'Customer Feedback': [\n        'Poor',\n        'unkown',\n        'Average',\n        'Good'\n    ],\n    'Education Level': [\n        \"Master's\",\n        \"PhD\",\n        \"Bachelor's\",\n        'High School'  \n        ],\n    'Policy Type':[\n        'Basic',\n        'Comprehensive',\n        'Premium'\n    ],\n    'Location':[\n        'Suburban',\n        'Rural', \n        'Urban',  \n    ],\n    'Exercise Frequency':[\n        'Monthly',\n        'Rarely', # Może zamienić motnhly i rarely\n        'Weekly',\n        'Daily'    \n    ]\n}\n    \n         ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:39:38.194191Z","iopub.execute_input":"2024-12-10T18:39:38.194649Z","iopub.status.idle":"2024-12-10T18:39:38.206692Z","shell.execute_reply.started":"2024-12-10T18:39:38.194602Z","shell.execute_reply":"2024-12-10T18:39:38.205615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:39:38.208179Z","iopub.execute_input":"2024-12-10T18:39:38.208591Z","iopub.status.idle":"2024-12-10T18:39:38.223571Z","shell.execute_reply.started":"2024-12-10T18:39:38.208546Z","shell.execute_reply":"2024-12-10T18:39:38.2225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder, OrdinalEncoder\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.metrics import accuracy_score\nfrom xgboost import XGBRegressor\nfrom sklearn.impute import SimpleImputer\n\n\nonehot_pipe = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='constant',fill_value='unkown')),\n    ('ordinal', OneHotEncoder(handle_unknown='ignore'))  \n])\n\nordinal_pipe = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='constant',fill_value='unkown')),\n    ('ordinal', OrdinalEncoder(categories=[ordinal_categories[col] for col in ordinal_cols]))\n])\n\nnumerical_pipe = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='constant',fill_value=0)),\n])\n\npreprocessor = ColumnTransformer(transformers=[\n    ('num', numerical_pipe, num_cols),\n    ('cat_ordinal', ordinal_pipe, ordinal_cols),\n    ('cat_onehot', onehot_pipe, one_hot_cols)\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:39:38.224936Z","iopub.execute_input":"2024-12-10T18:39:38.2253Z","iopub.status.idle":"2024-12-10T18:39:38.238795Z","shell.execute_reply.started":"2024-12-10T18:39:38.225268Z","shell.execute_reply":"2024-12-10T18:39:38.237627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBRegressor\n\nmodel = XGBRegressor(seed=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:39:38.425484Z","iopub.execute_input":"2024-12-10T18:39:38.425907Z","iopub.status.idle":"2024-12-10T18:39:38.431039Z","shell.execute_reply.started":"2024-12-10T18:39:38.42587Z","shell.execute_reply":"2024-12-10T18:39:38.429784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pipe = Pipeline(steps=[('preprocessor',preprocessor),('model',model)])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:39:38.89862Z","iopub.execute_input":"2024-12-10T18:39:38.899006Z","iopub.status.idle":"2024-12-10T18:39:38.904494Z","shell.execute_reply.started":"2024-12-10T18:39:38.898971Z","shell.execute_reply":"2024-12-10T18:39:38.903145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = df_train['Premium Amount']\nX = df_train.drop(columns=['Premium Amount'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:39:39.508707Z","iopub.execute_input":"2024-12-10T18:39:39.509269Z","iopub.status.idle":"2024-12-10T18:39:39.710066Z","shell.execute_reply.started":"2024-12-10T18:39:39.509215Z","shell.execute_reply":"2024-12-10T18:39:39.70889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.metrics import mean_squared_log_error\n\n\nkfold = KFold(n_splits=5, shuffle=True, random_state=42)\nrmsles = []\n\n\nfor train_index, val_index in kfold.split(X, y):\n    X_train, X_val = X.iloc[train_index], X.iloc[val_index]\n    y_train, y_val = y.iloc[train_index], y.iloc[val_index]\n\n    pipe.fit(X_train, y_train)\n\n    y_pred = pipe.predict(X_val)\n\n    y_pred = [max(single,20) for single in y_pred ]\n    rmsle = mean_squared_log_error(y_val, y_pred) ** (1/2)\n    rmsles.append(rmsle)\n\n    print(f\"Fold RMSLE: {rmsle:.4f}\")\n    \nprint(np.mean(rmsles))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:41:06.073204Z","iopub.execute_input":"2024-12-10T18:41:06.074796Z","iopub.status.idle":"2024-12-10T18:42:09.020548Z","shell.execute_reply.started":"2024-12-10T18:41:06.07475Z","shell.execute_reply":"2024-12-10T18:42:09.019516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pipe.fit(X,y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:42:18.704649Z","iopub.execute_input":"2024-12-10T18:42:18.705044Z","iopub.status.idle":"2024-12-10T18:42:44.755894Z","shell.execute_reply.started":"2024-12-10T18:42:18.705008Z","shell.execute_reply":"2024-12-10T18:42:44.754805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_df_test = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\nnew_df_test = new_df_test.drop(columns=['id'])\nnew_df_test = apply_feature_changes(new_df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:45:08.630222Z","iopub.execute_input":"2024-12-10T18:45:08.630632Z","iopub.status.idle":"2024-12-10T18:45:12.448988Z","shell.execute_reply.started":"2024-12-10T18:45:08.630597Z","shell.execute_reply":"2024-12-10T18:45:12.447955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test_pred = pipe.predict(new_df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:45:26.69813Z","iopub.execute_input":"2024-12-10T18:45:26.699087Z","iopub.status.idle":"2024-12-10T18:45:30.516968Z","shell.execute_reply.started":"2024-12-10T18:45:26.699024Z","shell.execute_reply":"2024-12-10T18:45:30.514722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_sample = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\") ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:46:47.751666Z","iopub.execute_input":"2024-12-10T18:46:47.752124Z","iopub.status.idle":"2024-12-10T18:46:47.950215Z","shell.execute_reply.started":"2024-12-10T18:46:47.752083Z","shell.execute_reply":"2024-12-10T18:46:47.949158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(submission_sample))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:46:53.794909Z","iopub.execute_input":"2024-12-10T18:46:53.795327Z","iopub.status.idle":"2024-12-10T18:46:53.800564Z","shell.execute_reply.started":"2024-12-10T18:46:53.795293Z","shell.execute_reply":"2024-12-10T18:46:53.799568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test_pred = [max(single,20) for single in y_test_pred]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:50:33.972162Z","iopub.execute_input":"2024-12-10T18:50:33.972601Z","iopub.status.idle":"2024-12-10T18:50:35.743009Z","shell.execute_reply.started":"2024-12-10T18:50:33.972564Z","shell.execute_reply":"2024-12-10T18:50:35.741774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_sample['Premium Amount'] = y_test_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:50:43.956728Z","iopub.execute_input":"2024-12-10T18:50:43.957115Z","iopub.status.idle":"2024-12-10T18:50:44.108775Z","shell.execute_reply.started":"2024-12-10T18:50:43.957083Z","shell.execute_reply":"2024-12-10T18:50:44.107607Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result = submission_sample.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:50:44.110366Z","iopub.execute_input":"2024-12-10T18:50:44.110677Z","iopub.status.idle":"2024-12-10T18:50:45.824735Z","shell.execute_reply.started":"2024-12-10T18:50:44.110647Z","shell.execute_reply":"2024-12-10T18:50:45.82355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_sample.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:50:45.826391Z","iopub.execute_input":"2024-12-10T18:50:45.82674Z","iopub.status.idle":"2024-12-10T18:50:45.836546Z","shell.execute_reply.started":"2024-12-10T18:50:45.826705Z","shell.execute_reply":"2024-12-10T18:50:45.835483Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_df_test = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:50:45.837702Z","iopub.execute_input":"2024-12-10T18:50:45.838093Z","iopub.status.idle":"2024-12-10T18:50:48.778892Z","shell.execute_reply.started":"2024-12-10T18:50:45.838031Z","shell.execute_reply":"2024-12-10T18:50:48.777935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sum(new_df_test['id'] != submission_sample['id'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:50:48.780888Z","iopub.execute_input":"2024-12-10T18:50:48.781255Z","iopub.status.idle":"2024-12-10T18:50:48.852642Z","shell.execute_reply.started":"2024-12-10T18:50:48.781222Z","shell.execute_reply":"2024-12-10T18:50:48.851444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(min(y_test_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T18:50:48.854346Z","iopub.execute_input":"2024-12-10T18:50:48.854804Z","iopub.status.idle":"2024-12-10T18:50:50.282662Z","shell.execute_reply.started":"2024-12-10T18:50:48.854726Z","shell.execute_reply":"2024-12-10T18:50:50.281378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}