{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"}],"dockerImageVersionId":30715,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I tried getting 50 MB of data but instead it provides 1.7 GB of data, i.e 100000 rows of data.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport gc  # Garbage Collector interface\n\n# Set the path to your CSV file\ncsv_file_path = '/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv'\n\n# Function to read the dataset in chunks and get data\ndef get_data(csv_file_path, chunk_size=50000):\n    total_size = 0\n    target_size = 50 * 1024 * 1024 \n    chunk_list = []  # list to hold chunks of data\n    \n    for chunk in pd.read_csv(csv_file_path, chunksize=chunk_size, engine='python'):\n        # Optionally convert data types here for memory efficiency\n        # Example: chunk['some_column'] = chunk['some_column'].astype('category')\n        \n        chunk_list.append(chunk)\n        total_size += chunk.memory_usage(deep=True).sum()\n        if total_size >= target_size:\n            break\n\n        # Clear list to free memory and force garbage collection\n        chunk_list = []\n        gc.collect()\n    \n    # Only store the final needed chunks\n    chunk_list.append(chunk)\n\n    # Concatenate all chunks to form the final dataframe\n    df_mb = pd.concat(chunk_list, ignore_index=True)\n    \n    # Free up memory\n    del chunk_list\n    gc.collect()\n    \n    return df_mb\n\n# Get data\ndf_mb = get_data(csv_file_path)\n\n# Save the data to a new CSV file\ndf_mb.to_csv('/kaggle/working/Leap_data.csv', index=False)\n\n# Display some information about the extracted data\nprint(f\"Extracted {df_mb.memory_usage(deep=True).sum() / (1024 * 1024):.2f} MB of data\")\nprint(df_mb.info())\n\nprint(\"Data has been saved to 'Leap_data.csv'\")\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}