{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"}],"dockerImageVersionId":30715,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport gc  # Garbage Collector interface\nimport os\nimport random\nfrom tqdm import tqdm\nimport dask.dataframe as dd","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-15T05:50:11.498524Z","iopub.execute_input":"2024-06-15T05:50:11.499078Z","iopub.status.idle":"2024-06-15T05:50:14.614131Z","shell.execute_reply.started":"2024-06-15T05:50:11.499037Z","shell.execute_reply":"2024-06-15T05:50:14.612527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Read a single data chunk from the training dataset, store it in a list, and finally write it to another CSV file.","metadata":{}},{"cell_type":"code","source":"# Function to read the dataset in chunks and get data\ndef get_data(csv_file_path):    \n    # Read csv as a list of blocks\n    ddf = dd.read_csv(csv_file_path, blocksize=f\"{1024}MB\") # 5 Gb of block size\n    for i in range(ddf.npartitions):\n#     for i in [0]:    \n        # Save the data to a new CSV file\n        folder = 'Leap_data_chunk'\n        file_name = f\"data_chunk_{i}.parquet\"\n        # Concatenate all chunks to form the final dataframe\n        df_mb = ddf.partitions[i]  \n        df_mb.to_parquet(folder, write_index=False,            # Don't write the index\n                         name_function=lambda x:file_name\n                        )\n\n        # Display some information about the extracted data\n        print(f\"Extracted {len(df_mb)} number of rows\")\n        print(df_mb.info())\n        print(f\"Data chunk has been saved to '{file_name}'\")\n        del df_mb\n        gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-06-15T05:50:14.616809Z","iopub.execute_input":"2024-06-15T05:50:14.617797Z","iopub.status.idle":"2024-06-15T05:50:14.629986Z","shell.execute_reply.started":"2024-06-15T05:50:14.617747Z","shell.execute_reply":"2024-06-15T05:50:14.62851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The path to CSV file\ncsv_file_path = '/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv'\n# Get data\nget_data(csv_file_path)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-06-15T05:50:14.632031Z","iopub.execute_input":"2024-06-15T05:50:14.632548Z","iopub.status.idle":"2024-06-15T05:51:07.020253Z","shell.execute_reply.started":"2024-06-15T05:50:14.632507Z","shell.execute_reply":"2024-06-15T05:51:07.018672Z"},"trusted":true},"execution_count":null,"outputs":[]}]}