{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%capture\n\nimport numpy as np\nimport pandas as pd\n!pip install netCDF4\nimport netCDF4\nfrom huggingface_hub import HfFileSystem\nimport xarray as xr\nimport polars as pl\nfrom huggingface_hub import hf_hub_download\nimport pickle\nfrom IPython.utils import io","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-28T13:41:06.287045Z","iopub.execute_input":"2024-05-28T13:41:06.287494Z","iopub.status.idle":"2024-05-28T13:41:26.513855Z","shell.execute_reply.started":"2024-05-28T13:41:06.287458Z","shell.execute_reply":"2024-05-28T13:41:26.512581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pl.read_csv('/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv', n_rows = 10)\n#train_df = pd.read_csv('/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv', float_precision='round_trip', nrows = 384)\n\nFEAT_COLS = train_df.columns[1:557]\nTARGET_COLS = train_df.columns[557:]\nX_cols = FEAT_COLS\ny_cols = TARGET_COLS\n\nX_cols_series_parents = ['state_t', 'state_q0001', 'state_q0002', 'state_q0003', 'state_u', 'state_v',\n                         'pbuf_ozone', 'pbuf_CH4', 'pbuf_N2O']\nX_cols_series = [x for x in X_cols if np.mean([x.startswith(y) for y in X_cols_series_parents])>0]\nX_cols_series_NOT = [x for x in X_cols if x not in X_cols_series]\nprint(len(X_cols_series)+len(X_cols_series_NOT) == len(X_cols))\nX_cols_series_indices = np.asarray([np.where(np.asarray(X_cols) == x)[0][0] for x in X_cols_series])\nX_cols_series_NOT_indices = np.asarray([np.where(np.asarray(X_cols) == x)[0][0] for x in X_cols_series_NOT])\n\ny_cols_series_parents = ['ptend_t', 'ptend_q0001', 'ptend_q0002', 'ptend_q0003', 'ptend_u', 'ptend_v']\ny_cols_series = [x for x in y_cols if np.mean([x.startswith(y) for y in y_cols_series_parents])>0]\ny_cols_series_NOT = [x for x in y_cols if x not in y_cols_series]\nprint(len(y_cols_series)+len(y_cols_series_NOT) == len(y_cols))\ny_cols_series_indices = np.asarray([np.where(np.asarray(y_cols) == x)[0][0] for x in y_cols_series])\ny_cols_series_NOT_indices = np.asarray([np.where(np.asarray(y_cols) == x)[0][0] for x in y_cols_series_NOT])","metadata":{"execution":{"iopub.status.busy":"2024-05-28T13:41:26.516221Z","iopub.execute_input":"2024-05-28T13:41:26.516985Z","iopub.status.idle":"2024-05-28T13:41:27.051322Z","shell.execute_reply.started":"2024-05-28T13:41:26.516942Z","shell.execute_reply":"2024-05-28T13:41:27.050166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfs = HfFileSystem()\nall_files = fs.glob(\"datasets/LEAP/ClimSim_low-res/train/*/E3SM-MMF.mli.*.nc\")\nprint(len(all_files)*384/7)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T13:41:27.052537Z","iopub.execute_input":"2024-05-28T13:41:27.052876Z","iopub.status.idle":"2024-05-28T13:43:21.042744Z","shell.execute_reply.started":"2024-05-28T13:41:27.052847Z","shell.execute_reply":"2024-05-28T13:43:21.041484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_data_cols(nc):\n    data_cols = []\n    for col in X_cols_series_parents:\n        arr = np.asarray(nc[col])\n        arr = np.transpose(arr, [1,0])\n        data_cols.append(arr)\n    data_cols = np.concatenate(data_cols, axis = -1)\n    return data_cols\n\ndef get_data_cols_not(nc):\n    data_cols_not = []\n    for col in X_cols_series_NOT:\n        arr = np.asarray(nc[col])\n        data_cols_not.append(arr)\n    data_cols_not = np.stack(data_cols_not, axis = -1)\n    return data_cols_not\n\ndef get_data_cols_target(nc, nc_target):\n    data_cols_targets = []\n    for col in X_cols_series_parents[:6]:\n        arr_in = np.asarray(nc[col])\n        arr_out = np.asarray(nc_target[col])\n        arr = (arr_out-arr_in)/1200\n        arr = np.transpose(arr, [1,0])\n        data_cols_targets.append(arr)\n    data_cols_targets = np.concatenate(data_cols_targets, axis = -1)\n    return data_cols_targets\n\ndef get_data_cols_not_target(nc_target):\n    data_cols_not_target = []\n    for col in y_cols_series_NOT:\n        arr = np.asarray(nc_target[col])\n        data_cols_not_target.append(arr)\n    data_cols_not_target = np.stack(data_cols_not_target, axis = -1)\n    return data_cols_not_target","metadata":{"execution":{"iopub.status.busy":"2024-05-28T13:43:21.045881Z","iopub.execute_input":"2024-05-28T13:43:21.046354Z","iopub.status.idle":"2024-05-28T13:43:21.058562Z","shell.execute_reply.started":"2024-05-28T13:43:21.046312Z","shell.execute_reply":"2024-05-28T13:43:21.057393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files = [all_files[i] for i in range(0, len(all_files), 32)]\nprint(len(files))","metadata":{"execution":{"iopub.status.busy":"2024-05-28T13:49:35.395552Z","iopub.execute_input":"2024-05-28T13:49:35.395946Z","iopub.status.idle":"2024-05-28T13:49:35.403287Z","shell.execute_reply.started":"2024-05-28T13:49:35.395917Z","shell.execute_reply":"2024-05-28T13:49:35.401844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_data_cols = np.zeros((384*len(files), 540))\nall_data_cols_not = np.zeros((384*len(files), 16))\nall_data_cols_targets = np.zeros((384*len(files), 360))\nall_data_cols_not_target = np.zeros((384*len(files), 8))\ntimes = []\nfor i, file_name in enumerate(files):\n    if i%10 == 0:\n        print(i)\n    file_name = file_name[30:]\n    with io.capture_output() as captured:\n        file = hf_hub_download(repo_id=\"LEAP/ClimSim_low-res\", filename=file_name, repo_type=\"dataset\");\n        file_target = hf_hub_download(repo_id=\"LEAP/ClimSim_low-res\", filename=file_name.replace('.mli.','.mlo.'), repo_type=\"dataset\");\n    nc = netCDF4.Dataset(file)\n    nc_target = netCDF4.Dataset(file_target)\n    \n    data_cols = get_data_cols(nc)\n    data_cols_not = get_data_cols_not(nc)\n    data_cols_targets = get_data_cols_target(nc, nc_target)\n    data_cols_not_target = get_data_cols_not_target(nc_target)\n    \n    all_data_cols[i*384:(i+1)*384, :] = data_cols\n    all_data_cols_not[i*384:(i+1)*384, :] = data_cols_not\n    all_data_cols_targets[i*384:(i+1)*384, :] = data_cols_targets\n    all_data_cols_not_target[i*384:(i+1)*384, :] = data_cols_not_target","metadata":{"execution":{"iopub.status.busy":"2024-05-28T13:49:37.192459Z","iopub.execute_input":"2024-05-28T13:49:37.192879Z","iopub.status.idle":"2024-05-28T13:49:40.343631Z","shell.execute_reply.started":"2024-05-28T13:49:37.192847Z","shell.execute_reply":"2024-05-28T13:49:40.34234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"times = [file_name[57:-3] for file_name in files]\nprint(len(times))\ntimes[0]","metadata":{"execution":{"iopub.status.busy":"2024-05-28T13:50:58.169399Z","iopub.execute_input":"2024-05-28T13:50:58.169863Z","iopub.status.idle":"2024-05-28T13:50:58.180944Z","shell.execute_reply.started":"2024-05-28T13:50:58.169828Z","shell.execute_reply":"2024-05-28T13:50:58.17945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(all_data_cols.shape)\nprint(all_data_cols_not.shape)\nprint(all_data_cols_targets.shape)\nprint(all_data_cols_not_target.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T13:51:04.938492Z","iopub.execute_input":"2024-05-28T13:51:04.938905Z","iopub.status.idle":"2024-05-28T13:51:04.945342Z","shell.execute_reply.started":"2024-05-28T13:51:04.938874Z","shell.execute_reply":"2024-05-28T13:51:04.943996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pickle.dump(times, open('times.p', 'bw'))\nnp.save('all_data_cols.npy', all_data_cols)\nnp.save('all_data_cols_not.npy', all_data_cols_not)\nnp.save('all_data_cols_targets.npy', all_data_cols_targets)\nnp.save('all_data_cols_not_target.npy', all_data_cols_not_target)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T13:51:08.871723Z","iopub.execute_input":"2024-05-28T13:51:08.872112Z","iopub.status.idle":"2024-05-28T13:51:08.899012Z","shell.execute_reply.started":"2024-05-28T13:51:08.872083Z","shell.execute_reply":"2024-05-28T13:51:08.897853Z"},"trusted":true},"execution_count":null,"outputs":[]}]}