{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-16T22:56:55.344565Z","iopub.execute_input":"2024-12-16T22:56:55.34491Z","iopub.status.idle":"2024-12-16T22:56:55.939871Z","shell.execute_reply.started":"2024-12-16T22:56:55.344876Z","shell.execute_reply":"2024-12-16T22:56:55.938646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T22:56:55.942327Z","iopub.execute_input":"2024-12-16T22:56:55.942959Z","iopub.status.idle":"2024-12-16T22:56:56.237951Z","shell.execute_reply.started":"2024-12-16T22:56:55.94291Z","shell.execute_reply":"2024-12-16T22:56:56.236495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pl.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T22:56:56.239738Z","iopub.execute_input":"2024-12-16T22:56:56.24094Z","iopub.status.idle":"2024-12-16T22:56:58.442336Z","shell.execute_reply.started":"2024-12-16T22:56:56.240892Z","shell.execute_reply":"2024-12-16T22:56:58.441358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T22:56:58.443495Z","iopub.execute_input":"2024-12-16T22:56:58.443847Z","iopub.status.idle":"2024-12-16T22:56:58.470869Z","shell.execute_reply.started":"2024-12-16T22:56:58.443806Z","shell.execute_reply":"2024-12-16T22:56:58.469665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T22:56:58.473543Z","iopub.execute_input":"2024-12-16T22:56:58.473935Z","iopub.status.idle":"2024-12-16T22:56:58.816129Z","shell.execute_reply.started":"2024-12-16T22:56:58.473894Z","shell.execute_reply":"2024-12-16T22:56:58.814855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Categories Dimensionality\ntrain_df.unpivot().group_by(\"variable\").n_unique().transpose()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T22:56:58.817413Z","iopub.execute_input":"2024-12-16T22:56:58.817839Z","iopub.status.idle":"2024-12-16T22:57:01.972926Z","shell.execute_reply.started":"2024-12-16T22:56:58.817793Z","shell.execute_reply":"2024-12-16T22:57:01.971663Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dropping","metadata":{}},{"cell_type":"code","source":"dropped_columns = [\"Policy Start Date\"]\ntrain_df = train_df.drop(dropped_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T22:57:01.974337Z","iopub.execute_input":"2024-12-16T22:57:01.974748Z","iopub.status.idle":"2024-12-16T22:57:01.981022Z","shell.execute_reply.started":"2024-12-16T22:57:01.974701Z","shell.execute_reply":"2024-12-16T22:57:01.979954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns = train_df.columns\nprint(f\"All columns: {', '.join(columns[:5])}...\\n\")\n\nnumerical_cols = \"Annual Income;id;Premium Amount;Credit Score;Health Score;Age;Previous Claims\".split(\";\")\nprint(f\"Numeric columns: {numerical_cols}\\n\")\nassert train_df[numerical_cols] is not None\n\ncategorical_cols = list(set(columns) - set(numerical_cols))\nprint(f\"Categorical columns: {categorical_cols}\")\n\nTARGET = \"Premium Amount\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T22:57:01.982153Z","iopub.execute_input":"2024-12-16T22:57:01.982464Z","iopub.status.idle":"2024-12-16T22:57:01.998201Z","shell.execute_reply.started":"2024-12-16T22:57:01.982433Z","shell.execute_reply":"2024-12-16T22:57:01.997126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Just categoricals\ntrain_df[categorical_cols].unpivot().group_by(\"variable\").n_unique().transpose()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T22:57:01.9998Z","iopub.execute_input":"2024-12-16T22:57:02.000229Z","iopub.status.idle":"2024-12-16T22:57:03.523258Z","shell.execute_reply.started":"2024-12-16T22:57:02.000185Z","shell.execute_reply":"2024-12-16T22:57:03.522129Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# X and y","metadata":{}},{"cell_type":"code","source":"X = train_df.drop(TARGET)\ny = train_df[TARGET]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T22:57:03.524607Z","iopub.execute_input":"2024-12-16T22:57:03.525022Z","iopub.status.idle":"2024-12-16T22:57:03.53089Z","shell.execute_reply.started":"2024-12-16T22:57:03.524979Z","shell.execute_reply":"2024-12-16T22:57:03.529828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\nfrom sklearn.preprocessing import OneHotEncoder\n\nXy_encoded = X.with_columns(X[categorical_cols].to_dummies()).drop(categorical_cols).with_columns(y)\n\nXy_encoded.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T23:15:22.462716Z","iopub.execute_input":"2024-12-16T23:15:22.463892Z","iopub.status.idle":"2024-12-16T23:15:23.003651Z","shell.execute_reply.started":"2024-12-16T23:15:22.463833Z","shell.execute_reply":"2024-12-16T23:15:23.002589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Xy_encoded[TARGET] is not None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T23:15:31.724186Z","iopub.execute_input":"2024-12-16T23:15:31.724561Z","iopub.status.idle":"2024-12-16T23:15:31.731408Z","shell.execute_reply.started":"2024-12-16T23:15:31.72453Z","shell.execute_reply":"2024-12-16T23:15:31.730247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Xy_encoded.shape[0] * Xy_encoded.shape[1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T23:15:42.623741Z","iopub.execute_input":"2024-12-16T23:15:42.624164Z","iopub.status.idle":"2024-12-16T23:15:42.631026Z","shell.execute_reply.started":"2024-12-16T23:15:42.62413Z","shell.execute_reply":"2024-12-16T23:15:42.629942Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train / Val","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_data, val_data = train_test_split(Xy_encoded, test_size=0.1, random_state=42)\ntrain_data.shape, val_data.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T23:15:46.409301Z","iopub.execute_input":"2024-12-16T23:15:46.410115Z","iopub.status.idle":"2024-12-16T23:15:46.783879Z","shell.execute_reply.started":"2024-12-16T23:15:46.410071Z","shell.execute_reply":"2024-12-16T23:15:46.782822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install lightning -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T22:57:22.844934Z","iopub.execute_input":"2024-12-16T22:57:22.845321Z","iopub.status.idle":"2024-12-16T22:57:34.228343Z","shell.execute_reply.started":"2024-12-16T22:57:22.84529Z","shell.execute_reply":"2024-12-16T22:57:34.227056Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Le Model","metadata":{}},{"cell_type":"code","source":"f\"Input Size: {X_encoded.shape[1]}\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T22:57:34.230708Z","iopub.execute_input":"2024-12-16T22:57:34.231108Z","iopub.status.idle":"2024-12-16T22:57:34.238535Z","shell.execute_reply.started":"2024-12-16T22:57:34.231074Z","shell.execute_reply":"2024-12-16T22:57:34.237417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom torch import optim, nn, utils, Tensor\nfrom torchvision.datasets import MNIST\nfrom torchvision.transforms import ToTensor\nimport lightning as L\nfrom lightning.pytorch.utilities import grad_norm\n\n# define any number of nn.Modules (or use your current ones)\nregression_network = nn.Sequential(\n    nn.Linear(X_encoded.shape[1], 64), \n    nn.ReLU(),\n    nn.Linear(64, 32),\n    nn.ReLU(),\n    nn.Linear(32, 1)\n)\n\n# define the LightningModule\nclass NeuralNetwork(L.LightningModule):\n    def __init__(self, regression_network):\n        super().__init__()\n        self.network = regression_network\n\n    def training_step(self, batch, batch_idx):\n        # training_step defines the train loop.\n        # it is independent of forward\n        x, y_true = batch[:, :-1], batch[:, -1]\n        y_pred = self.network(x).flatten()\n        loss = nn.functional.mse_loss(y_pred, y_true)\n        \n        print(f\"train_loss: {loss} (pred ~= true => {y_pred[0]} ~= {y_true[0]})\", loss)\n        return loss\n\n    def configure_optimizers(self):\n        optimizer = optim.Adam(self.parameters(), lr=1e-3)\n        return optimizer\n\n    def on_before_optimizer_step(self, optimizer):\n        # Compute the 2-norm for each layer\n        # If using mixed precision, the gradients are already unscaled here\n        norms = grad_norm(self.network, norm_type=2)\n        print(norms)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T23:23:16.887183Z","iopub.execute_input":"2024-12-16T23:23:16.88806Z","iopub.status.idle":"2024-12-16T23:23:16.89935Z","shell.execute_reply.started":"2024-12-16T23:23:16.888016Z","shell.execute_reply":"2024-12-16T23:23:16.898188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch.utils.data import DataLoader\nimport torch\n\nBATCH_SIZE = 512\n\nmodel = NeuralNetwork(regression_network)\n\ntrain_tensor = train_data.to_torch(dtype=pl.Float32)\nval_tensor = val_data.to_torch(dtype=pl.Float32)\n\ntrain_dataloader = DataLoader(train_tensor, batch_size=BATCH_SIZE)\nval_dataloader = DataLoader(val_tensor, batch_size=BATCH_SIZE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T23:23:17.23593Z","iopub.execute_input":"2024-12-16T23:23:17.236329Z","iopub.status.idle":"2024-12-16T23:23:17.500986Z","shell.execute_reply.started":"2024-12-16T23:23:17.236294Z","shell.execute_reply":"2024-12-16T23:23:17.499883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trainer = L.Trainer(limit_train_batches=100, max_epochs=1)\ntrainer.fit(model=model, train_dataloaders=train_dataloader)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T23:23:17.978392Z","iopub.execute_input":"2024-12-16T23:23:17.979236Z","iopub.status.idle":"2024-12-16T23:23:19.161164Z","shell.execute_reply.started":"2024-12-16T23:23:17.979196Z","shell.execute_reply":"2024-12-16T23:23:19.160085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}