{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import dask.dataframe as dd\n\n# Definición de columnas para input y test\ntemporal_cols_input = [f'{col}_{i}' for col in ['state_t', 'state_q0001', 'state_q0002', 'state_q0003', 'state_u', 'pbuf_ozone', 'pbuf_CH4', 'pbuf_N2O'] for i in range(60)]\nnon_temporal_cols_input = ['state_ps', 'pbuf_SOLIN', 'pbuf_LHFLX', 'pbuf_SHFLX', 'pbuf_TAUX', 'pbuf_TAUY', 'pbuf_COSZRS', 'cam_in_ALDIF', 'cam_in_ALDIR', 'cam_in_ASDIF', 'cam_in_ASDIR', 'cam_in_LWUP', 'cam_in_ICEFRAC', 'cam_in_LANDFRAC', 'cam_in_OCNFRAC', 'cam_in_SNOWHLAND']\n# Definición de columnas para target\ntemporal_cols_target = [f'{col}_{i}' for col in ['ptend_t', 'ptend_q0001', 'ptend_q0002', 'ptend_q0003', 'ptend_u', 'ptend_v'] for i in range(60)]\nnon_temporal_cols_target = ['cam_out_NETSW', 'cam_out_FLWDS', 'cam_out_PRECSC', 'cam_out_PRECC', 'cam_out_SOLS', 'cam_out_SOLL', 'cam_out_SOLSD', 'cam_out_SOLLD']\n\n# Rutas de los archivos CSV\ninput_path = '/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv'\ntest_path = '/kaggle/input/leap-atmospheric-physics-ai-climsim/test.csv'\ntarget_path = '/kaggle/input/leap-atmospheric-physics-ai-climsim/sample_submission.csv'\nchunksize = '256MB' # Puedes ajustar esto según tu hardware\n\ndef eliminar_extremos(texto, num_inicio=4, num_fin=4):\n    \"\"\"Elimina los primeros y últimos caracteres de una cadena de texto.\"\"\"\n    return texto[num_inicio:-num_fin]\n\ndef create_parquet(file_path, columns):\n    \"\"\"Cargar los datos en chunks y limitar el DataFrame a NUM_SAMPLES filas.\"\"\"\n    # Leer datos con Dask DataFrame\n    ddf = dd.read_csv(file_path, usecols=columns, blocksize=chunksize)\n    \n    # Crear un nombre adecuado para el archivo parquet\n    nombre_archivo = eliminar_extremos(file_path, num_inicio=4, num_fin=4)\n    temp_file = f'/kaggle/input/leap-atmospheric-physics-ai-climsim/{nombre_archivo}-full.parquet'\n    \n    # Guardar el DataFrame limitado como archivo en formato Parquet\n    ddf.to_parquet(temp_file)","metadata":{"execution":{"iopub.status.busy":"2024-05-25T04:09:31.678351Z","iopub.execute_input":"2024-05-25T04:09:31.678836Z","iopub.status.idle":"2024-05-25T04:09:31.692138Z","shell.execute_reply.started":"2024-05-25T04:09:31.678801Z","shell.execute_reply":"2024-05-25T04:09:31.690321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# crear parquet con datos de train\ncreate_parquet(input_path, temporal_cols_input + non_temporal_cols_input)","metadata":{"execution":{"iopub.status.busy":"2024-05-25T03:46:27.919019Z","iopub.execute_input":"2024-05-25T03:46:27.919418Z","iopub.status.idle":"2024-05-25T04:07:26.110181Z","shell.execute_reply.started":"2024-05-25T03:46:27.919387Z","shell.execute_reply":"2024-05-25T04:07:26.107837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# crear parquet con datos de test\ncreate_parquet(test_path, temporal_cols_input + non_temporal_cols_input)","metadata":{"execution":{"iopub.status.busy":"2024-05-25T04:09:39.505781Z","iopub.execute_input":"2024-05-25T04:09:39.506293Z","iopub.status.idle":"2024-05-25T04:09:40.123948Z","shell.execute_reply.started":"2024-05-25T04:09:39.506251Z","shell.execute_reply":"2024-05-25T04:09:40.122058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# crear parquet con datos de target\ncreate_parquet(target_path, temporal_cols_target + non_temporal_cols_target)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{},"execution_count":null,"outputs":[]}]}