{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **P4E12 - Data Visualization and Neural Network Predictions with Tensorflow**","metadata":{}},{"cell_type":"markdown","source":"# Version History","metadata":{}},{"cell_type":"markdown","source":"* Version 1\n\n Initialization\n\n* Version 2\n\nIncreases epochs 10 --> 25\n\nAdded dropout regulazation\n\n* Version 3\n\nAdded histograms and pie charts for features to view distrubution\n\nImplemented cyclic encoding to datetime data\n\nInspired by [@habilamar](https://www.kaggle.com/habilamar)'s notebook, [KG S4E12](https://www.kaggle.com/code/habilamar/kg-s4e12/notebook)","metadata":{}},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"#Data Cleaning\nimport pandas as pd\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import MinMaxScaler\n\n#Data Visualization\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n#Neural Network with TensorFlow\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense\nfrom sklearn.model_selection import train_test_split\n\n#Submission\nimport pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:46:39.074129Z","iopub.execute_input":"2024-12-13T04:46:39.074595Z","iopub.status.idle":"2024-12-13T04:46:39.081512Z","shell.execute_reply.started":"2024-12-13T04:46:39.074556Z","shell.execute_reply":"2024-12-13T04:46:39.08031Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Cleaning","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import MinMaxScaler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:46:39.083587Z","iopub.execute_input":"2024-12-13T04:46:39.084049Z","iopub.status.idle":"2024-12-13T04:46:39.097017Z","shell.execute_reply.started":"2024-12-13T04:46:39.084003Z","shell.execute_reply":"2024-12-13T04:46:39.095733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\ntrain.info(verbose=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:46:39.098786Z","iopub.execute_input":"2024-12-13T04:46:39.099155Z","iopub.status.idle":"2024-12-13T04:46:47.214393Z","shell.execute_reply.started":"2024-12-13T04:46:39.09912Z","shell.execute_reply":"2024-12-13T04:46:47.213225Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"First, we have to transform the date columns into seperate columns for the year, month, day, and day of the week.","metadata":{}},{"cell_type":"code","source":"train['Policy Start Date'] = pd.to_datetime(train['Policy Start Date'])\ntrain['Policy Start Year'] = train['Policy Start Date'].dt.year\ntrain['Policy Start Month'] = train['Policy Start Date'].dt.month\ntrain['Policy Start Day'] = train['Policy Start Date'].dt.day\ntrain['Policy Start DayOfWeek'] = train['Policy Start Date'].dt.dayofweek\ntrain.drop('Policy Start Date',axis=1,inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:46:47.216525Z","iopub.execute_input":"2024-12-13T04:46:47.216897Z","iopub.status.idle":"2024-12-13T04:46:48.098117Z","shell.execute_reply.started":"2024-12-13T04:46:47.216862Z","shell.execute_reply":"2024-12-13T04:46:48.096748Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"In the next box, we are just going to create various diffent dataframes for diffrent purposes","metadata":{}},{"cell_type":"code","source":"#only the column with the values we are trying to predict\ny = train['Premium Amount']\n\n#only the columns with the features\nx = train.drop(columns=['Premium Amount','id'])\n\n#catagorical data\nx_cat= x.select_dtypes(include=['object'])\n\n#numerical data\nx_nums = x.select_dtypes(exclude=['object'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:46:48.099537Z","iopub.execute_input":"2024-12-13T04:46:48.099891Z","iopub.status.idle":"2024-12-13T04:46:48.721489Z","shell.execute_reply.started":"2024-12-13T04:46:48.099858Z","shell.execute_reply":"2024-12-13T04:46:48.72031Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"First, we are going to clean the data of any null values","metadata":{}},{"cell_type":"code","source":"#this function is borrowed from my other projects. Feel free to use it\ndef clean_NaN(train_mini, strategy='mean'):\n    try:\n        for column in train_mini:\n            imputer=SimpleImputer(strategy=strategy)\n            train_mini[column]=imputer.fit_transform(train_mini[column].values.reshape(-1,1))\n    except:\n        imputer=SimpleImputer(strategy=strategy)\n        train_mini=imputer.fit_transform(train_mini.values.reshape(-1,1))\n    return train_mini","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:46:48.724452Z","iopub.execute_input":"2024-12-13T04:46:48.72495Z","iopub.status.idle":"2024-12-13T04:46:48.731537Z","shell.execute_reply.started":"2024-12-13T04:46:48.724899Z","shell.execute_reply":"2024-12-13T04:46:48.730459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_nums = clean_NaN(x_nums, 'median')\nx_nums.info(verbose=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:46:48.733418Z","iopub.execute_input":"2024-12-13T04:46:48.733873Z","iopub.status.idle":"2024-12-13T04:46:50.724862Z","shell.execute_reply.started":"2024-12-13T04:46:48.73382Z","shell.execute_reply":"2024-12-13T04:46:50.723783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in x_cat:\n    cats = x_cat[col].nunique()\n    print(f\"'{col}' has {cats} unique categories.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:46:50.726313Z","iopub.execute_input":"2024-12-13T04:46:50.726733Z","iopub.status.idle":"2024-12-13T04:46:51.350111Z","shell.execute_reply.started":"2024-12-13T04:46:50.726693Z","shell.execute_reply":"2024-12-13T04:46:51.348808Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Since there is only a few catagories in each column, we will convert the catagories to numerical data using one hot encoding.","metadata":{}},{"cell_type":"code","source":"x_dummies = pd.get_dummies(x_cat, columns=(x_cat.columns), dummy_na=True, dtype='float64')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:46:51.351523Z","iopub.execute_input":"2024-12-13T04:46:51.351851Z","iopub.status.idle":"2024-12-13T04:46:52.796582Z","shell.execute_reply.started":"2024-12-13T04:46:51.35182Z","shell.execute_reply":"2024-12-13T04:46:52.79527Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_dummies.info(verbose=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:46:52.798261Z","iopub.execute_input":"2024-12-13T04:46:52.798776Z","iopub.status.idle":"2024-12-13T04:46:52.810857Z","shell.execute_reply.started":"2024-12-13T04:46:52.798724Z","shell.execute_reply":"2024-12-13T04:46:52.809626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#if you want to see every column\nx_dummies.info(verbose=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:46:52.812468Z","iopub.execute_input":"2024-12-13T04:46:52.812854Z","iopub.status.idle":"2024-12-13T04:46:52.925084Z","shell.execute_reply.started":"2024-12-13T04:46:52.812819Z","shell.execute_reply":"2024-12-13T04:46:52.923775Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"One hot encoding basiclly creates new columns for each catagory","metadata":{}},{"cell_type":"code","source":"x_cat = x_cat.astype('category')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:46:52.928916Z","iopub.execute_input":"2024-12-13T04:46:52.929301Z","iopub.status.idle":"2024-12-13T04:46:53.761774Z","shell.execute_reply.started":"2024-12-13T04:46:52.929267Z","shell.execute_reply":"2024-12-13T04:46:53.760215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.concat([x_nums,x_dummies,y],axis=1)\ntrain_mini = pd.concat([x_nums,x_cat,y],axis=1)\nx = pd.concat([x_nums, x_dummies], axis=1)","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-13T04:46:53.76348Z","iopub.execute_input":"2024-12-13T04:46:53.76383Z","iopub.status.idle":"2024-12-13T04:46:55.000753Z","shell.execute_reply.started":"2024-12-13T04:46:53.763799Z","shell.execute_reply":"2024-12-13T04:46:54.999193Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Visualization","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom math import ceil","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:46:55.002228Z","iopub.execute_input":"2024-12-13T04:46:55.002744Z","iopub.status.idle":"2024-12-13T04:46:55.008039Z","shell.execute_reply.started":"2024-12-13T04:46:55.002682Z","shell.execute_reply":"2024-12-13T04:46:55.006812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(y.describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:46:55.009768Z","iopub.execute_input":"2024-12-13T04:46:55.010245Z","iopub.status.idle":"2024-12-13T04:46:55.078832Z","shell.execute_reply.started":"2024-12-13T04:46:55.010193Z","shell.execute_reply":"2024-12-13T04:46:55.076927Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"With a histogram, we can see how the target variable is distrubuted. We can see that our target variable has a simple, curved slope downward and is right-skewed.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nplt.hist(y, bins=50, color='skyblue', edgecolor='black')\nplt.title('Premium Amounts')\nplt.xlabel('Amount')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:46:55.08098Z","iopub.execute_input":"2024-12-13T04:46:55.081343Z","iopub.status.idle":"2024-12-13T04:46:55.434277Z","shell.execute_reply.started":"2024-12-13T04:46:55.081303Z","shell.execute_reply":"2024-12-13T04:46:55.432852Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We can check the distrubution for the rest of the features. For numerical data, we use a histogram and for catagorical, we use a pie chart.","metadata":{}},{"cell_type":"code","source":"columns = int(2)\nrows = int(ceil(len(x_nums.columns)/columns))\n_, ax = plt.subplots(rows, columns, figsize=(30,20*columns))\nfor i in range(len(x_nums.columns)):\n    sns.histplot(x=train[x_nums.columns[i]],kde=True,ax=ax[i//columns,i%columns])\n    ax[i//columns,i%columns].set_xlabel(x_nums.columns[i], fontsize=32)\n\ncolumns = int(2)\nrows = int(ceil(len(x_cat.columns)/columns))\n_, ax = plt.subplots(rows, columns, figsize=(30,20*columns))\nfor i in range(len(x_cat.columns)):\n    unique,counts=np.unique(train_mini[x_cat.columns[i]].astype(str),return_counts=True)\n    ax[i//columns,i%columns].pie(x=counts,labels=unique,autopct='%.0f%%',textprops={'fontsize': 22})\n    ax[i//columns,i%columns].set_xlabel(x_cat.columns[i], fontsize=32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:46:55.435845Z","iopub.execute_input":"2024-12-13T04:46:55.436301Z","iopub.status.idle":"2024-12-13T04:48:11.545254Z","shell.execute_reply.started":"2024-12-13T04:46:55.436251Z","shell.execute_reply":"2024-12-13T04:48:11.544135Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"A correlation_matrix can help someone see how each indivual feature relates to each other. We can use this to find the most important features.","metadata":{}},{"cell_type":"code","source":"correlation_matrix = train_mini.corr(numeric_only=True)\nplt.figure(figsize=(12, 8))\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm', fmt='.2f')\nplt.title('Correlation Matrix')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:48:11.546599Z","iopub.execute_input":"2024-12-13T04:48:11.546917Z","iopub.status.idle":"2024-12-13T04:48:13.002734Z","shell.execute_reply.started":"2024-12-13T04:48:11.546886Z","shell.execute_reply":"2024-12-13T04:48:13.00161Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This is rough. No feature has a significant relation to anything.","metadata":{}},{"cell_type":"markdown","source":"Below we look more closely at the relationship. For numerical data, we use a hexmap which changes color based on how many points are in the area. For catagorical data, we use a boxplot.","metadata":{}},{"cell_type":"code","source":"for col in x_nums.columns:\n    plt.figure(figsize=(10, 6))\n    plt.hexbin(train[col], train['Premium Amount'], gridsize=30, cmap='GnBu')\n    plt.colorbar(label='Point Density')\n    plt.title(f'{col} vs Target Variable')\n    plt.xlabel(col)\n    plt.ylabel('Premium Amount')\n    plt.show()\n\nfor col in x_cat.columns:\n    plt.figure(figsize=(12, 6))\n    categories = train_mini[col].dropna().unique()  # Get unique categories\n    data = [train_mini[train_mini[col] == category]['Premium Amount'].values for category in categories]\n    plt.boxplot(data, labels=categories, patch_artist=True, boxprops=dict(facecolor='lightblue'),medianprops=dict(color='black', linewidth=3))\n    plt.title(f'{col} vs Premium Amount')\n    plt.xlabel(col)\n    plt.ylabel('Premium Amount')\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:48:13.004161Z","iopub.execute_input":"2024-12-13T04:48:13.004524Z","iopub.status.idle":"2024-12-13T04:48:23.87829Z","shell.execute_reply.started":"2024-12-13T04:48:13.004489Z","shell.execute_reply":"2024-12-13T04:48:23.876201Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As we saw from the correlation chart, there is no visible relationship.","metadata":{}},{"cell_type":"markdown","source":"We also normalize the data. This makes the data between -1 and 1, which makes it easier to process than if it was over a wider range.","metadata":{}},{"cell_type":"code","source":"#this function is borrowed from my other projects. Feel free to use it\ndef normalize(table,columns):\n    scaler = MinMaxScaler()\n    \n    for column in columns:\n        table[column] = scaler.fit_transform(table[column].values.reshape(-1,1))\n\nnormalize(x, x_nums.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:48:23.880208Z","iopub.execute_input":"2024-12-13T04:48:23.880655Z","iopub.status.idle":"2024-12-13T04:48:24.008099Z","shell.execute_reply.started":"2024-12-13T04:48:23.880615Z","shell.execute_reply":"2024-12-13T04:48:24.00688Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Lastly, we convert the datetime features using cylical feature encoding. Normally, time features will repeat and go back to 0. For example, the days of the week go 0, 1, 2, 3, 4, 5, 6, 0... Cylical encoding makes this more curved, which machine learning models can understand better.","metadata":{}},{"cell_type":"code","source":"datetimes = ['Policy Start Year','Policy Start Month','Policy Start Day','Policy Start DayOfWeek']\n\nfor col in datetimes:\n    x[col + '_sin'] = np.sin(2 * np.pi * x[col]/x[col].max())\n    x[col + '_cos'] = np.cos(2 * np.pi * x[col]/x[col].max())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:48:24.009805Z","iopub.execute_input":"2024-12-13T04:48:24.010173Z","iopub.status.idle":"2024-12-13T04:48:24.301506Z","shell.execute_reply.started":"2024-12-13T04:48:24.010139Z","shell.execute_reply":"2024-12-13T04:48:24.300573Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Neural Network with TensorFlow","metadata":{}},{"cell_type":"markdown","source":"For complex relationships, I like to use neural networks. One of the best packages to make a neural network is Tensorflow.","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:48:24.302901Z","iopub.execute_input":"2024-12-13T04:48:24.303357Z","iopub.status.idle":"2024-12-13T04:48:24.309991Z","shell.execute_reply.started":"2024-12-13T04:48:24.303289Z","shell.execute_reply":"2024-12-13T04:48:24.308762Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"First we have to split the data into a train and val set. The val set is used to test the model on data it hasn't seen before.","metadata":{}},{"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(x, y, test_size=0.2)#, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:48:24.311537Z","iopub.execute_input":"2024-12-13T04:48:24.312043Z","iopub.status.idle":"2024-12-13T04:48:25.352804Z","shell.execute_reply.started":"2024-12-13T04:48:24.312007Z","shell.execute_reply":"2024-12-13T04:48:25.351576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = Sequential([\n    Dense(256, activation='relu'), \n    Dense(128, activation='relu'),\n    Dense(64, activation='relu'),\n    Dense(32, activation='relu'),\n    Dropout(0.1),\n    Dense(1)\n])\n\nmodel.compile(optimizer='adam', loss='mean_squared_logarithmic_error')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:48:25.354248Z","iopub.execute_input":"2024-12-13T04:48:25.35465Z","iopub.status.idle":"2024-12-13T04:48:25.375855Z","shell.execute_reply.started":"2024-12-13T04:48:25.354613Z","shell.execute_reply":"2024-12-13T04:48:25.374635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(x_train, y_train, epochs= 25, batch_size=128, validation_data=(x_val, y_val)) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:51:03.070176Z","iopub.execute_input":"2024-12-13T04:51:03.070606Z","iopub.status.idle":"2024-12-13T04:53:17.13322Z","shell.execute_reply.started":"2024-12-13T04:51:03.070569Z","shell.execute_reply":"2024-12-13T04:53:17.131636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Model Loss Over Epochs')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend(loc='upper right')\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:53:17.13593Z","iopub.execute_input":"2024-12-13T04:53:17.136313Z","iopub.status.idle":"2024-12-13T04:53:17.410527Z","shell.execute_reply.started":"2024-12-13T04:53:17.136278Z","shell.execute_reply":"2024-12-13T04:53:17.409303Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"markdown","source":"First, we have to transform the test data set the same we transformed the train.","metadata":{}},{"cell_type":"code","source":"ids = test['id']\ntest = test.drop(columns=['id'])\n\ntest['Policy Start Date'] = pd.to_datetime(test['Policy Start Date'])\ntest['Policy Start Year'] = test['Policy Start Date'].dt.year\ntest['Policy Start Month'] = test['Policy Start Date'].dt.month\ntest['Policy Start Day'] = test['Policy Start Date'].dt.day\ntest['Policy Start DayOfWeek'] = test['Policy Start Date'].dt.dayofweek\ntest.drop('Policy Start Date',axis=1,inplace=True)\n\ntest_cat= test.select_dtypes(include=['object'])\ntest_nums = test.select_dtypes(exclude=['object'])\nprint(\"Finished splitting dataframe\")\n\ntest_nums = clean_NaN(test_nums, 'median')\nnormalize(test_nums, test_nums.columns)\nprint('Finished proccesing numerical data')\n\ntest_dummies = pd.get_dummies(test_cat, columns=(test_cat.columns), dummy_na=True, dtype='float64')\nprint('Finished proccesing catagorical data')\n\ntest = pd.concat([test_nums, test_dummies], axis=1)\nfor col in datetimes:\n    test[col + '_sin'] = np.sin(2 * np.pi * test[col]/test[col].max())\n    test[col + '_cos'] = np.cos(2 * np.pi * test[col]/test[col].max())\nprint('Finished combining dataframes')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:53:17.41175Z","iopub.execute_input":"2024-12-13T04:53:17.412054Z","iopub.status.idle":"2024-12-13T04:53:21.330284Z","shell.execute_reply.started":"2024-12-13T04:53:17.412022Z","shell.execute_reply":"2024-12-13T04:53:21.329147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = model.predict(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:53:26.891229Z","iopub.execute_input":"2024-12-13T04:53:26.891607Z","iopub.status.idle":"2024-12-13T04:54:09.813069Z","shell.execute_reply.started":"2024-12-13T04:53:26.89157Z","shell.execute_reply":"2024-12-13T04:54:09.811609Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = submission.flatten()\n\nsubmission = pd.DataFrame({'id' : ids, 'Premium Amount' : submission})\nsubmission.to_csv('submission.csv', index=False)\n\nprint(submission)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T04:54:09.815812Z","iopub.execute_input":"2024-12-13T04:54:09.816493Z","iopub.status.idle":"2024-12-13T04:54:11.129512Z","shell.execute_reply.started":"2024-12-13T04:54:09.816413Z","shell.execute_reply":"2024-12-13T04:54:11.128269Z"}},"outputs":[],"execution_count":null}]}