{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":56537,"databundleVersionId":8877088,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-28T16:16:22.58708Z","iopub.execute_input":"2024-06-28T16:16:22.587517Z","iopub.status.idle":"2024-06-28T16:16:23.156585Z","shell.execute_reply.started":"2024-06-28T16:16:22.587486Z","shell.execute_reply":"2024-06-28T16:16:23.155067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import polars as pl\nimport numpy as np\nimport seaborn as sns\nimport sklearn as sk\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nimport torch\nfrom torch.utils.data import DataLoader, TensorDataset\n","metadata":{"execution":{"iopub.status.busy":"2024-06-28T16:16:23.158823Z","iopub.execute_input":"2024-06-28T16:16:23.159383Z","iopub.status.idle":"2024-06-28T16:16:28.065415Z","shell.execute_reply.started":"2024-06-28T16:16:23.159349Z","shell.execute_reply":"2024-06-28T16:16:28.063932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nneeded_columns = [\n    \"^ptend_q0001.*$\", \"^ptend_q0002.*$\", \"^ptend_q0003.*$\",\n    \"^state_q0001.*$\", \"^state_q0002.*$\", \"^state_q0003.*$\",\n    \"^cam_out_PRECC.*$\", \"^state_ps.*$\",\"^cam_out_PRECSC.*$\"\n]\n\n\nleap_raw_df = pl.scan_csv('/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv')\nbig = leap_raw_df.select(pl.col(c) for c in needed_columns).fetch(750000)\n\ngravidade = 9.80665\n\ntendencia_vapor = big.select(pl.col(\"^ptend_q0001.*$\")).to_numpy()\ntendencia_liquido = big.select(pl.col(\"^ptend_q0002.*$\")).to_numpy()\ntendencia_gelo = big.select(pl.col(\"^ptend_q0003.*$\")).to_numpy()\nvapor = big.select(pl.col(\"^state_q0001.*$\")).to_numpy()\nliquido = big.select(pl.col(\"^state_q0002.*$\")).to_numpy()\ngelo = big.select(pl.col(\"^state_q0003.*$\")).to_numpy()\nchuva = big.select(pl.col(\"^cam_out_PRECC.*$\")).to_numpy().flatten()\nneve = big.select(pl.col(\"^cam_out_PRECSC.*$\")).to_numpy().flatten()\npressao = big.select(pl.col(\"^state_ps.*$\")).to_numpy().flatten()\n\n# Vectorized operations to calculate volume_inicial and volume_predito\nvolume_inicial = np.sum(vapor + liquido + gelo, axis=1) * (pressao / gravidade)\nvolume_predito = (np.sum(tendencia_vapor + tendencia_liquido + tendencia_gelo, axis=1) * (pressao / gravidade)) + chuva + neve\n\n# Calculate error and water_error flag\nerro = abs((volume_predito - volume_inicial) / volume_inicial +1)\nwater_error = (erro > 0.00005).astype(np.int8)\n\n# Create DataFrame with erro and water_error columns\ndf = pl.DataFrame({\n    \"erro\": erro,\n    \"water_error\": water_error\n})\n\n# Get the maximum error value\nmax_erro = df[\"erro\"].max()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-28T16:16:28.067796Z","iopub.execute_input":"2024-06-28T16:16:28.068396Z","iopub.status.idle":"2024-06-28T16:17:37.880973Z","shell.execute_reply.started":"2024-06-28T16:16:28.068357Z","shell.execute_reply":"2024-06-28T16:17:37.878377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum(df['water_error']==1)","metadata":{"execution":{"iopub.status.busy":"2024-06-28T16:17:37.883598Z","iopub.execute_input":"2024-06-28T16:17:37.884039Z","iopub.status.idle":"2024-06-28T16:17:37.962707Z","shell.execute_reply.started":"2024-06-28T16:17:37.884004Z","shell.execute_reply":"2024-06-28T16:17:37.961372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(df[\"erro\"])","metadata":{"execution":{"iopub.status.busy":"2024-06-28T16:17:37.965355Z","iopub.execute_input":"2024-06-28T16:17:37.965766Z","iopub.status.idle":"2024-06-28T16:17:41.866118Z","shell.execute_reply.started":"2024-06-28T16:17:37.965734Z","shell.execute_reply":"2024-06-28T16:17:41.864592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_columns = leap_raw_df.columns[1:556]\noutput_columns = leap_raw_df.columns[556:924]","metadata":{"execution":{"iopub.status.busy":"2024-06-28T16:17:41.867663Z","iopub.execute_input":"2024-06-28T16:17:41.868129Z","iopub.status.idle":"2024-06-28T16:17:41.898443Z","shell.execute_reply.started":"2024-06-28T16:17:41.86809Z","shell.execute_reply":"2024-06-28T16:17:41.89704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chuva = big.select(pl.col(\"^cam_out_PRECC.*$\"))\nneve = big.select(pl.col(\"^cam_out_PRECSC.*$\"))\nplt.boxplot(chuva.filter(pl.col(\"cam_out_PRECC\") > 0.0000001))\n","metadata":{"execution":{"iopub.status.busy":"2024-06-28T16:17:41.899892Z","iopub.execute_input":"2024-06-28T16:17:41.900299Z","iopub.status.idle":"2024-06-28T16:17:42.187116Z","shell.execute_reply.started":"2024-06-28T16:17:41.900268Z","shell.execute_reply":"2024-06-28T16:17:42.18578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.boxplot(neve.filter(pl.col(\"cam_out_PRECSC\") > 0))","metadata":{"execution":{"iopub.status.busy":"2024-06-28T16:17:42.188588Z","iopub.execute_input":"2024-06-28T16:17:42.18896Z","iopub.status.idle":"2024-06-28T16:17:42.468356Z","shell.execute_reply.started":"2024-06-28T16:17:42.188928Z","shell.execute_reply":"2024-06-28T16:17:42.46723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"big.describe()","metadata":{"execution":{"iopub.status.busy":"2024-06-28T13:32:24.618692Z","iopub.status.idle":"2024-06-28T13:32:24.619125Z","shell.execute_reply.started":"2024-06-28T13:32:24.618926Z","shell.execute_reply":"2024-06-28T13:32:24.618944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def adicionar_agua_v_erro_info(big: pl.DataFrame, margen = 0.00005) ->pl.DataFrame:\n    gravidade = 9.80665\n    ptend_vapor = big.select(pl.col(\"^ptend_q0001.*$\")).to_numpy()\n    ptend_liquido = big.select(pl.col(\"^ptend_q0002.*$\")).to_numpy()\n    ptend_gelo = big.select(pl.col(\"^ptend_q0003.*$\")).to_numpy()\n    \n    state_vapor = big.select(pl.col(\"^state_q0001.*$\")).to_numpy()\n    state_liquido = big.select(pl.col(\"^state_q0002.*$\")).to_numpy()\n    state_gelo = big.select(pl.col(\"^state_q0003.*$\")).to_numpy()\n    \n    precipitacao_agua = big.select(pl.col(\"^cam_out_PRECC.*$\")).to_numpy().flatten()\n    precipitacao_neve = big.select(pl.col(\"^cam_out_PRECSC.*$\")).to_numpy().flatten()\n    \n    pressao = big.select(pl.col(\"^state_ps.*$\")).to_numpy().flatten()\n    volume_inicial = np.sum(state_vapor + state_liquido + state_gelo,axis=1)*(pressao/gravidade)\n    volume_predito = np.sum(ptend_vapor + ptend_liquido + ptend_gelo,axis=1)*(pressao/gravidade) + chuva + neve\n    \n    \n    erro_volumetrico = abs((volume_predito - volume_inicial)/volume_inicial +1)\n    flag_erro_volumetrico = (erro >margen).astype(np.int8)\n    big = big.with_columns([\n        pl.Series(\"erro_volumetrico\",erro_volumetrico),\n        pl.Series(\"flag_erro_volumetrico\",flag_erro_volumetrico)\n    ])\n    return big","metadata":{"execution":{"iopub.status.busy":"2024-06-28T13:32:24.620794Z","iopub.status.idle":"2024-06-28T13:32:24.621264Z","shell.execute_reply.started":"2024-06-28T13:32:24.621056Z","shell.execute_reply":"2024-06-28T13:32:24.621074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}