{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport shutil\nimport pandas as pd\nfrom tqdm.auto import tqdm\nfrom pathlib import Path","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-22T18:16:41.309001Z","iopub.execute_input":"2023-06-22T18:16:41.309412Z","iopub.status.idle":"2023-06-22T18:16:41.44284Z","shell.execute_reply.started":"2023-06-22T18:16:41.309377Z","shell.execute_reply":"2023-06-22T18:16:41.441562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/asl-fingerspelling/train_landmarks\"\n\nFEATURES_TO_LOAD = (\n    [f\"x_left_hand_{i}\" for i in range(20)] +\n    [f\"y_left_hand_{i}\" for i in range(20)] +\n    [f\"x_right_hand_{i}\" for i in range(20)] +\n    [f\"y_right_hand_{i}\" for i in range(20)]\n)","metadata":{"execution":{"iopub.status.busy":"2023-06-22T18:16:41.444809Z","iopub.execute_input":"2023-06-22T18:16:41.446001Z","iopub.status.idle":"2023-06-22T18:16:41.452213Z","shell.execute_reply.started":"2023-06-22T18:16:41.445962Z","shell.execute_reply":"2023-06-22T18:16:41.451096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_parquet_files = list(Path(BASE_PATH).rglob(\"*.parquet\"))\nall_parquet_files[:2], len(all_parquet_files)","metadata":{"execution":{"iopub.status.busy":"2023-06-22T18:16:41.453542Z","iopub.execute_input":"2023-06-22T18:16:41.453908Z","iopub.status.idle":"2023-06-22T18:16:41.480648Z","shell.execute_reply.started":"2023-06-22T18:16:41.453876Z","shell.execute_reply":"2023-06-22T18:16:41.479642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing steps:\n* reads all parquet files, \n* extracts the hand features: `FEATURES_TO_LOAD`, \n* round them to the 3rd decimal place\n* split into unique files based on `sequence_id`\n* save each file into its own csv","metadata":{}},{"cell_type":"code","source":"for file in all_parquet_files:\n    df = pd.read_parquet(file)\n    folder = f\"dataset/{file.stem}\"\n    os.makedirs(folder, exist_ok=True)\n    for sequence_id, sub_df in tqdm(df.groupby(level=0)):\n        sub_df[FEATURES_TO_LOAD].to_csv(f\"{folder}/{sequence_id}.csv\", index=False, float_format='%.3f')\n    print(f\"Processed file {file.stem}.\")","metadata":{"execution":{"iopub.status.busy":"2023-06-22T18:19:40.96314Z","iopub.execute_input":"2023-06-22T18:19:40.96358Z","iopub.status.idle":"2023-06-22T18:20:19.443179Z","shell.execute_reply.started":"2023-06-22T18:19:40.963546Z","shell.execute_reply":"2023-06-22T18:20:19.44196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Compressing datasets\")\nshutil.make_archive(\"dataset\", 'zip', \"dataset\")\nshutil.rmtree(\"dataset\") ","metadata":{"execution":{"iopub.status.busy":"2023-06-22T18:20:22.148548Z","iopub.execute_input":"2023-06-22T18:20:22.148997Z","iopub.status.idle":"2023-06-22T18:20:28.070394Z","shell.execute_reply.started":"2023-06-22T18:20:22.148963Z","shell.execute_reply":"2023-06-22T18:20:28.069239Z"},"trusted":true},"execution_count":null,"outputs":[]}]}