{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **FORWARD**\n\n- This is a starter data set preparation kernel that helps to prepare a smaller version of the train dataset for one to at least commence the process. <br>\n- We start off with a **smaller subset of the train data with small batches of 5k rows per batch** and save the files in parquet format <br>\n- All files are saved with the label - **Train_Batch{i}.parquet** with the Batch id i indicating the batch number. Each table batch is approximately 35 Mb large and hosts 5k rows individually <br>\n- We envisage reading the data in small batches on purpose so that the user could choose batches based on his/ her system requirements. Once the user determines the adequate number of batches for his/ her requirements, he/ she may concatenate them suitably. <br>\n- As an example, if a user concatenates batches 1-10, he is likely to read in 5000 * 10 = 50_000 rows <br>\n","metadata":{}},{"cell_type":"code","source":"import polars as pl\nimport polars.selectors as cs\nfrom gc import collect\n\nimport ctypes;\nlibc = ctypes.CDLL(\"libc.so.6\")\nimport os","metadata":{"execution":{"iopub.status.busy":"2024-04-20T09:28:36.734905Z","iopub.execute_input":"2024-04-20T09:28:36.735272Z","iopub.status.idle":"2024-04-20T09:28:37.038949Z","shell.execute_reply.started":"2024-04-20T09:28:36.735244Z","shell.execute_reply":"2024-04-20T09:28:37.037854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **DATA PREPARATION**","metadata":{}},{"cell_type":"code","source":"%%time \n\n# User may enter his/ her desired batch size and chunk size:-\nbatch_size  = 5000\nnb_batches  = 500\nstrt_row_nb = 5_000_000\n\nprint(f\"---> Total rows read = {(batch_size * nb_batches)/1_000_000 :,.2f} million\");\nprint(f\"---> Reading after rows {strt_row_nb} from the table header\");\n\nreader = pl.read_csv_batched(\"/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv\",\n                             separator              = \",\",\n                             try_parse_dates        = True,\n                             batch_size             = batch_size ,\n                             skip_rows_after_header = strt_row_nb\n                            )  \n\nprint();\nbatches = reader.next_batches(nb_batches)  \nfor i, df in enumerate(batches): \n    df = df.with_columns(cs.all().shrink_dtype())\n    df.write_parquet(f\"Train_Batch{i}.parquet\")\n    est_size  = df.estimated_size(unit = \"mb\")\n    rows_read = batch_size * (i+1) / 1000\n    print(f\"---> Estimated size = {est_size :.2f} Mb | Rows read = {rows_read}k | Batch {i}\")\n    del df, est_size, rows_read;\n    \n    collect();\n    libc.malloc_trim(0);","metadata":{"execution":{"iopub.status.busy":"2024-04-20T09:29:07.206661Z","iopub.execute_input":"2024-04-20T09:29:07.20708Z","iopub.status.idle":"2024-04-20T09:29:08.900625Z","shell.execute_reply.started":"2024-04-20T09:29:07.207047Z","shell.execute_reply":"2024-04-20T09:29:08.899442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nprint(f\"\\nEstimating the total size of the working folder after all batches are stored\\n\")\nsize   = 0\nmypath = '/kaggle/working'\n\nfor i in os.scandir(mypath):\n    size+=os.path.getsize(i)\nprint(f\"Folder size = {size / 1_000_000_000:,.4f} GB\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}