{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"}],"dockerImageVersionId":30699,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch \nimport torch.nn as nn \nimport warnings\n\n# Suppress FutureWarning messages\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nimport os\nimport polars as pl\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:32:45.570332Z","iopub.execute_input":"2024-05-04T08:32:45.570881Z","iopub.status.idle":"2024-05-04T08:32:50.388091Z","shell.execute_reply.started":"2024-05-04T08:32:45.570843Z","shell.execute_reply":"2024-05-04T08:32:50.387198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = pl.read_csv_batched('/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv', batch_size=62500)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:32:50.390278Z","iopub.execute_input":"2024-05-04T08:32:50.39132Z","iopub.status.idle":"2024-05-04T08:32:50.454231Z","shell.execute_reply.started":"2024-05-04T08:32:50.391284Z","shell.execute_reply":"2024-05-04T08:32:50.453142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading a single Batch of training samples into data variable\ndata = train_df.next_batches(1)[0]    ","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:32:50.455677Z","iopub.execute_input":"2024-05-04T08:32:50.456458Z","iopub.status.idle":"2024-05-04T08:33:05.402564Z","shell.execute_reply.started":"2024-05-04T08:32:50.456418Z","shell.execute_reply":"2024-05-04T08:33:05.401717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Variable Names for Feature and Prediction Columns\nFEAT_COLS = data.columns[1:557]\nTARGET_COLS = data.columns[557:]","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:33:05.404191Z","iopub.execute_input":"2024-05-04T08:33:05.404492Z","iopub.status.idle":"2024-05-04T08:33:05.411909Z","shell.execute_reply.started":"2024-05-04T08:33:05.404468Z","shell.execute_reply":"2024-05-04T08:33:05.410944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fixed_targets = ['ptend_q0001_4', 'ptend_q0001_5', 'ptend_q0001_6', 'ptend_q0001_7',\n                 'ptend_q0001_8', 'ptend_q0001_9', 'ptend_q0001_10', 'ptend_q0001_11',\n                 'ptend_q0002_0', 'ptend_q0002_1', 'ptend_q0002_2', 'ptend_q0002_3',\n                 'ptend_q0002_4', 'ptend_q0002_5', 'ptend_q0002_6', 'ptend_q0002_7',\n                 'ptend_q0002_8', 'ptend_q0002_9', 'ptend_q0002_10', 'ptend_q0002_11',\n                 'ptend_q0002_12', 'ptend_q0002_13', 'ptend_q0002_14', 'ptend_q0002_15',\n                 'ptend_q0002_16', 'ptend_q0002_17', 'ptend_q0002_18', 'ptend_q0002_19',\n                 'ptend_q0002_20', 'ptend_q0002_21', 'ptend_q0002_22', 'ptend_q0002_23', \n                 'ptend_q0003_0', 'ptend_q0003_1', 'ptend_q0003_2', 'ptend_q0003_3',\n                 'ptend_q0003_4', 'ptend_q0003_5', 'ptend_q0003_6', 'ptend_q0003_7',\n                 'ptend_q0003_8', 'ptend_q0003_9', 'ptend_q0003_10', 'ptend_q0003_11',\n                 'ptend_u_0', 'ptend_u_1', 'ptend_u_2', 'ptend_u_3', 'ptend_u_4',\n                 'ptend_u_5', 'ptend_u_6', 'ptend_u_7', 'ptend_u_8', 'ptend_u_9',\n                 'ptend_u_10', 'ptend_u_11', 'ptend_v_0', 'ptend_v_1', 'ptend_v_2',\n                 'ptend_v_3', 'ptend_v_4', 'ptend_v_5', 'ptend_v_6', 'ptend_v_7',\n                 'ptend_v_8', 'ptend_v_9', 'ptend_v_10', 'ptend_v_11']\n\nX = data.select(FEAT_COLS)\ny = data.select(TARGET_COLS)\n\n# Converting Data from F64 to F32 Format\nfor col in FEAT_COLS:\n    X = X.with_columns(pl.col(col).cast(pl.Float32))\nfor col in TARGET_COLS:\n    y = y.with_columns(pl.col(col).cast(pl.Float32))\n    \ny_fixed = y.select(fixed_targets)\ny = y.drop(fixed_targets)\n\n# To be used during Submission\ny_columns = y.columns\ny_fixed_columns = y_fixed.columns","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:33:26.610711Z","iopub.execute_input":"2024-05-04T08:33:26.611672Z","iopub.status.idle":"2024-05-04T08:33:27.255708Z","shell.execute_reply.started":"2024-05-04T08:33:26.611634Z","shell.execute_reply":"2024-05-04T08:33:27.254516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_fixed.shape)\nprint(y.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:33:29.69136Z","iopub.execute_input":"2024-05-04T08:33:29.691718Z","iopub.status.idle":"2024-05-04T08:33:29.696692Z","shell.execute_reply.started":"2024-05-04T08:33:29.691688Z","shell.execute_reply":"2024-05-04T08:33:29.695795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Converting Data into numpy format and splitting into train and test data\nX, y = X.to_numpy(), y.to_numpy()\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.1)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:33:31.422661Z","iopub.execute_input":"2024-05-04T08:33:31.423256Z","iopub.status.idle":"2024-05-04T08:33:32.322965Z","shell.execute_reply.started":"2024-05-04T08:33:31.423209Z","shell.execute_reply":"2024-05-04T08:33:32.32204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train.shape)\nprint(y_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:33:32.889158Z","iopub.execute_input":"2024-05-04T08:33:32.889635Z","iopub.status.idle":"2024-05-04T08:33:32.894289Z","shell.execute_reply.started":"2024-05-04T08:33:32.889605Z","shell.execute_reply":"2024-05-04T08:33:32.893433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_test.shape)\nprint(y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:33:33.236677Z","iopub.execute_input":"2024-05-04T08:33:33.236984Z","iopub.status.idle":"2024-05-04T08:33:33.24135Z","shell.execute_reply.started":"2024-05-04T08:33:33.236958Z","shell.execute_reply":"2024-05-04T08:33:33.240471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pre-Processing - Deleting Feature Columns with 0 or almost 0 standard deviation \ndeleted_columns = np.where(np.std(X_train, axis=0) < 1e-6)[0]\nX_train_cleaned = np.delete(X_train, deleted_columns, 1)\nprint(X_train_cleaned.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:33:34.5993Z","iopub.execute_input":"2024-05-04T08:33:34.599647Z","iopub.status.idle":"2024-05-04T08:33:34.769769Z","shell.execute_reply.started":"2024-05-04T08:33:34.59962Z","shell.execute_reply":"2024-05-04T08:33:34.76883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pre-Processing - Standard Scaling our Cleaned Features\nfrom sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nscaler.fit(X_train_cleaned)\nX_train_preprocessed = scaler.transform(X_train_cleaned)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:33:35.716936Z","iopub.execute_input":"2024-05-04T08:33:35.71779Z","iopub.status.idle":"2024-05-04T08:33:35.914877Z","shell.execute_reply.started":"2024-05-04T08:33:35.717761Z","shell.execute_reply":"2024-05-04T08:33:35.914016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining our XGBRegressor model to be computed using GPU\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.multioutput import MultiOutputRegressor\n\nmodel = XGBRegressor(tree_method='hist', device=\"cuda\")\nmulti_model = MultiOutputRegressor(model)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:34:01.847556Z","iopub.execute_input":"2024-05-04T08:34:01.8479Z","iopub.status.idle":"2024-05-04T08:34:02.011883Z","shell.execute_reply.started":"2024-05-04T08:34:01.84787Z","shell.execute_reply":"2024-05-04T08:34:02.011179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmulti_model.fit(X_train_preprocessed, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:34:03.29925Z","iopub.execute_input":"2024-05-04T08:34:03.299656Z","iopub.status.idle":"2024-05-04T08:44:33.66493Z","shell.execute_reply.started":"2024-05-04T08:34:03.299627Z","shell.execute_reply":"2024-05-04T08:44:33.663981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pre-Processing our X_test data and calculating predictions using trained multi_model\nX_test_cleaned = np.delete(X_test, deleted_columns, 1)\nX_test_preprocessed = scaler.transform(X_test_cleaned)\npredictions = multi_model.predict(X_test_preprocessed)\npredictions","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:44:43.596036Z","iopub.execute_input":"2024-05-04T08:44:43.596399Z","iopub.status.idle":"2024-05-04T08:44:53.719463Z","shell.execute_reply.started":"2024-05-04T08:44:43.59637Z","shell.execute_reply":"2024-05-04T08:44:53.718504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting our score for Test Dataset\nmulti_model.score(X_test_preprocessed,y_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:44:57.524971Z","iopub.execute_input":"2024-05-04T08:44:57.525885Z","iopub.status.idle":"2024-05-04T08:45:07.319856Z","shell.execute_reply.started":"2024-05-04T08:44:57.525845Z","shell.execute_reply":"2024-05-04T08:45:07.318871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting our score for Train Dataset\nmulti_model.score(X_train_preprocessed, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:45:09.559658Z","iopub.execute_input":"2024-05-04T08:45:09.560235Z","iopub.status.idle":"2024-05-04T08:47:11.184081Z","shell.execute_reply.started":"2024-05-04T08:45:09.560201Z","shell.execute_reply":"2024-05-04T08:47:11.183135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load Test Sample Features\ntest_df = pl.read_csv('/kaggle/input/leap-atmospheric-physics-ai-climsim/test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:47:28.322775Z","iopub.execute_input":"2024-05-04T08:47:28.323567Z","iopub.status.idle":"2024-05-04T08:47:53.182417Z","shell.execute_reply.started":"2024-05-04T08:47:28.323535Z","shell.execute_reply":"2024-05-04T08:47:53.181413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in FEAT_COLS:\n    test_dataframe = test_df.select(FEAT_COLS).with_columns(pl.col(col).cast(pl.Float32))\n\ntest_dataframe = test_dataframe.to_numpy()\n\nprint(test_dataframe.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:48:11.368156Z","iopub.execute_input":"2024-05-04T08:48:11.369074Z","iopub.status.idle":"2024-05-04T08:48:15.665608Z","shell.execute_reply.started":"2024-05-04T08:48:11.36904Z","shell.execute_reply":"2024-05-04T08:48:15.664726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pre-Processing our Test Samples\ntest_dataframe_cleaned = np.delete(test_dataframe, deleted_columns, 1)\ntest_dataframe_preprocessed = scaler.transform(test_dataframe_cleaned)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:48:17.247568Z","iopub.execute_input":"2024-05-04T08:48:17.247914Z","iopub.status.idle":"2024-05-04T08:48:18.717381Z","shell.execute_reply.started":"2024-05-04T08:48:17.247885Z","shell.execute_reply":"2024-05-04T08:48:18.716587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting the output from our trained model for the test samples\npredictions = multi_model.predict(test_dataframe_preprocessed)\npredictions","metadata":{"execution":{"iopub.status.busy":"2024-05-04T08:48:20.536586Z","iopub.execute_input":"2024-05-04T08:48:20.537157Z","iopub.status.idle":"2024-05-04T09:09:59.965724Z","shell.execute_reply.started":"2024-05-04T08:48:20.537128Z","shell.execute_reply":"2024-05-04T09:09:59.964614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading submission File\nimport pandas as pd\nsub = pd.read_csv(\"/kaggle/input/leap-atmospheric-physics-ai-climsim/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-05-04T09:10:08.063433Z","iopub.execute_input":"2024-05-04T09:10:08.064282Z","iopub.status.idle":"2024-05-04T09:11:32.762097Z","shell.execute_reply.started":"2024-05-04T09:10:08.064248Z","shell.execute_reply":"2024-05-04T09:11:32.761277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-04T09:11:45.223273Z","iopub.execute_input":"2024-05-04T09:11:45.223913Z","iopub.status.idle":"2024-05-04T09:11:45.259145Z","shell.execute_reply.started":"2024-05-04T09:11:45.223881Z","shell.execute_reply":"2024-05-04T09:11:45.258301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using the submission sample target values as weights \nsub.loc[:,y_columns] *= predictions","metadata":{"execution":{"iopub.status.busy":"2024-05-04T09:11:47.80935Z","iopub.execute_input":"2024-05-04T09:11:47.809955Z","iopub.status.idle":"2024-05-04T09:11:49.366478Z","shell.execute_reply.started":"2024-05-04T09:11:47.809926Z","shell.execute_reply":"2024-05-04T09:11:49.365681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.loc[:, y_fixed_columns] *= np.mean(y_fixed.to_numpy(), axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T09:11:59.704042Z","iopub.execute_input":"2024-05-04T09:11:59.70509Z","iopub.status.idle":"2024-05-04T09:12:00.087186Z","shell.execute_reply.started":"2024-05-04T09:11:59.70505Z","shell.execute_reply":"2024-05-04T09:12:00.086368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-04T09:12:02.364076Z","iopub.execute_input":"2024-05-04T09:12:02.364907Z","iopub.status.idle":"2024-05-04T09:12:02.389278Z","shell.execute_reply.started":"2024-05-04T09:12:02.364873Z","shell.execute_reply":"2024-05-04T09:12:02.388302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(sub.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T09:12:05.121262Z","iopub.execute_input":"2024-05-04T09:12:05.121785Z","iopub.status.idle":"2024-05-04T09:12:05.126723Z","shell.execute_reply.started":"2024-05-04T09:12:05.121749Z","shell.execute_reply":"2024-05-04T09:12:05.125772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Outputting our submission file\ntest_polars = pl.from_pandas(sub[[\"sample_id\"]+TARGET_COLS])\ntest_polars.write_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-05-04T09:12:08.613158Z","iopub.execute_input":"2024-05-04T09:12:08.613525Z","iopub.status.idle":"2024-05-04T09:12:22.017916Z","shell.execute_reply.started":"2024-05-04T09:12:08.613495Z","shell.execute_reply":"2024-05-04T09:12:22.016886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}