{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport polars as pl\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-23T06:14:13.653727Z","iopub.execute_input":"2024-04-23T06:14:13.654179Z","iopub.status.idle":"2024-04-23T06:14:15.166642Z","shell.execute_reply.started":"2024-04-23T06:14:13.654141Z","shell.execute_reply":"2024-04-23T06:14:15.165262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_path='/kaggle/input/leap-atmospheric-physics-ai-climsim/'","metadata":{"execution":{"iopub.status.busy":"2024-04-23T06:14:15.169055Z","iopub.execute_input":"2024-04-23T06:14:15.169603Z","iopub.status.idle":"2024-04-23T06:14:15.175634Z","shell.execute_reply.started":"2024-04-23T06:14:15.169557Z","shell.execute_reply":"2024-04-23T06:14:15.174211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_13_pbuf_N2O = [\"pbuf_N2O_\"+str(i) for i in range(27,60)]\ncols_13_pbuf_CH4 = [\"pbuf_CH4_\"+str(i) for i in range(27,60)]\n\ncols_to_delete = cols_13_pbuf_N2O + cols_13_pbuf_CH4\n\ncols_to_delete.append('sample_id')\nlen(cols_to_delete)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T06:14:15.177031Z","iopub.execute_input":"2024-04-23T06:14:15.177408Z","iopub.status.idle":"2024-04-23T06:14:15.196815Z","shell.execute_reply.started":"2024-04-23T06:14:15.177377Z","shell.execute_reply":"2024-04-23T06:14:15.195434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df=pd.read_csv(data_path+'sample_submission.csv',nrows=2)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T06:14:15.198492Z","iopub.execute_input":"2024-04-23T06:14:15.198938Z","iopub.status.idle":"2024-04-23T06:14:15.247215Z","shell.execute_reply.started":"2024-04-23T06:14:15.198861Z","shell.execute_reply":"2024-04-23T06:14:15.245951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.drop(columns='sample_id',inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T06:14:15.250394Z","iopub.execute_input":"2024-04-23T06:14:15.250768Z","iopub.status.idle":"2024-04-23T06:14:15.263892Z","shell.execute_reply.started":"2024-04-23T06:14:15.250737Z","shell.execute_reply":"2024-04-23T06:14:15.262731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_cols=sub_df.columns","metadata":{"execution":{"iopub.status.busy":"2024-04-23T06:14:15.265336Z","iopub.execute_input":"2024-04-23T06:14:15.265729Z","iopub.status.idle":"2024-04-23T06:14:15.276271Z","shell.execute_reply.started":"2024-04-23T06:14:15.265698Z","shell.execute_reply":"2024-04-23T06:14:15.275007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train=pl.read_csv('/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv',n_rows=100000)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T06:14:15.277624Z","iopub.execute_input":"2024-04-23T06:14:15.27806Z","iopub.status.idle":"2024-04-23T06:14:28.507009Z","shell.execute_reply.started":"2024-04-23T06:14:15.278025Z","shell.execute_reply":"2024-04-23T06:14:28.505466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test=pl.read_csv('/kaggle/input/leap-atmospheric-physics-ai-climsim/test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-23T06:14:28.509162Z","iopub.execute_input":"2024-04-23T06:14:28.509655Z","iopub.status.idle":"2024-04-23T06:15:17.098958Z","shell.execute_reply.started":"2024-04-23T06:14:28.509596Z","shell.execute_reply":"2024-04-23T06:15:17.097593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data_chunk=pd.read_csv(data_path+'train.csv',chunksize=100000,usecols=lambda col: col not in cols_to_delete)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T06:15:17.100508Z","iopub.execute_input":"2024-04-23T06:15:17.100864Z","iopub.status.idle":"2024-04-23T06:15:17.113815Z","shell.execute_reply.started":"2024-04-23T06:15:17.100837Z","shell.execute_reply":"2024-04-23T06:15:17.112439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n# from sklearn.linear_model import LinearRegression\n# from sklearn.metrics import r2_score\n# from sklearn.decomposition import PCA\n# import pickle\n\n# # Initialize list to store R^2 scores\n# r2_scores = []\n\n# # Iterate through each chunk of data\n# for chunk_idx, chunk in enumerate(data_chunk, 1):\n#     # Split chunk into features (X_train) and target (y_train)\n#     y_train = chunk[target_cols]\n#     X_train = chunk.drop(columns=target_cols)\n    \n#     # Apply PCA to X_train\n#     pca = PCA()\n#     X_train_pca = pca.fit_transform(X_train)\n    \n#     # Train a linear regression model on PCA-transformed data\n#     model_pca = LinearRegression()\n#     model_pca.fit(X_train_pca, y_train)\n    \n#     # Calculate R^2 score\n#     y_pred_pca = model_pca.predict(X_train_pca)\n#     r2_pca = r2_score(y_train, y_pred_pca)\n    \n#     # Print and store R^2 score\n#     print(f\"Chunk {chunk_idx} - R^2 Score: {r2_pca}\")\n#     r2_scores.append(r2_pca)\n    \n#     # Write R^2 score to pickle file\n#     r2_filename = f'r2_score_chunk_{chunk_idx}.pkl'\n#     with open(r2_filename, 'wb') as r2_file:\n#         pickle.dump(r2_pca, r2_file)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T06:15:17.115615Z","iopub.execute_input":"2024-04-23T06:15:17.116275Z","iopub.status.idle":"2024-04-23T06:15:17.12397Z","shell.execute_reply.started":"2024-04-23T06:15:17.116226Z","shell.execute_reply":"2024-04-23T06:15:17.122496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train=data_train.drop(columns='sample_id')","metadata":{"execution":{"iopub.status.busy":"2024-04-23T06:15:17.125832Z","iopub.execute_input":"2024-04-23T06:15:17.12627Z","iopub.status.idle":"2024-04-23T06:15:17.175026Z","shell.execute_reply.started":"2024-04-23T06:15:17.126235Z","shell.execute_reply":"2024-04-23T06:15:17.173459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test=data_test.drop(columns='sample_id')","metadata":{"execution":{"iopub.status.busy":"2024-04-23T06:15:17.176494Z","iopub.execute_input":"2024-04-23T06:15:17.176873Z","iopub.status.idle":"2024-04-23T06:15:17.184227Z","shell.execute_reply.started":"2024-04-23T06:15:17.176839Z","shell.execute_reply":"2024-04-23T06:15:17.18265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import polars as pl\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import r2_score\n\n\n# Extract features (X_train) and target (y_train) from data_train\nX_train = data_train.drop(target_cols)\ny_train = data_train.select(target_cols)  # Assuming target_cols has only one element\n\n# Fit a simple linear regression model\nmodel = LinearRegression()\nmodel.fit(X_train.to_numpy(), y_train.to_numpy())\n\n# Predict the target variable for data_train and calculate R^2 score\ny_pred_train = model.predict(X_train.to_numpy())\nr2_train = r2_score(y_train.to_numpy(), y_pred_train)\n\nprint(\"R^2 score on data_train:\", r2_train)\n\n# Predict the target variable for data_test\ny_pred_test = model.predict(data_test.to_numpy())\n\n# You can now use y_pred_test for further analysis or evaluation\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T06:15:17.185888Z","iopub.execute_input":"2024-04-23T06:15:17.186435Z","iopub.status.idle":"2024-04-23T06:15:40.9115Z","shell.execute_reply.started":"2024-04-23T06:15:17.186397Z","shell.execute_reply":"2024-04-23T06:15:40.909872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss = pd.read_csv(\"/kaggle/input/leap-atmospheric-physics-ai-climsim/sample_submission.csv\")\nss.iloc[:,1:] *= y_pred_test\nss.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T06:15:40.9153Z","iopub.execute_input":"2024-04-23T06:15:40.915723Z","iopub.status.idle":"2024-04-23T06:18:29.788715Z","shell.execute_reply.started":"2024-04-23T06:15:40.91569Z","shell.execute_reply":"2024-04-23T06:18:29.786638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n# from sklearn.metrics import r2_score\n# from xgboost import XGBRegressor\n\n# # Initialize lists to store R^2 scores and feature coefficients\n# r2_scores = []\n# feature_importance = []\n\n# # Initialize an empty list to store data frames\n# results_data_frames = []\n\n# # Iterate through each chunk of data\n# for chunk in data_chunk:\n#     selected_columns = [col for col in chunk.columns if 'pbuf_ozone' in col or 'pbuf_N20' in col]\n#     # Split chunk into features (X_train) and target (y_train)\n#     y_train = chunk[target_cols]\n#     X_train = chunk.drop(columns=target_cols)\n# #     X_train = X_train[selected_columns]  # Select relevant columns\n    \n#     # Train an XGBoost regressor model\n#     model = XGBRegressor()\n#     model.fit(X_train, y_train)\n\n#     # Calculate R^2 score\n#     y_pred = model.predict(X_train)\n#     r2 = r2_score(y_train, y_pred)\n#     r2_scores.append(r2)\n\n#     # Extract feature importances\n#     feature_importance.append(model.feature_importances_)\n    \n#     print(\"Chunk\")\n#     print(\"R^2 Score:\", r2)\n# #     print(\"Feature Importance:\")\n# #     # Print feature importances\n# #     for feature, importance in zip(X_train.columns, model.feature_importances_):\n# #         print(f\"{feature}: {importance}\")\n# #     print()\n\n# # No need to create data frames or store results when using XGBoost\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T06:18:29.789834Z","iopub.status.idle":"2024-04-23T06:18:29.790287Z","shell.execute_reply.started":"2024-04-23T06:18:29.790094Z","shell.execute_reply":"2024-04-23T06:18:29.790112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}