{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Data Description\n**Files**\n- `index.json`: a dict of index-character pairs\n- `train.npy`: train numpy data, a `(F,75,3)` shape numpy object array\n    - `F` stand for the amount of total frames in train dataset\n    - `75` is the amount of skeleton points(without face points)\n    - `3` is the `[x,y,z]` coordinates\n- `train.pickle`: train labels, a dict\n    1. `label_list`: a list of labels, each label is a array of character's index\n    2. `sequence_id`: a list of sequence_id.\n    3. `start_list`: a list of each sequence index frames\n    4. `length_list`: a list of each sequence length","metadata":{}},{"cell_type":"markdown","source":"# Config","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-22T10:13:56.980393Z","iopub.execute_input":"2023-05-22T10:13:56.980816Z","iopub.status.idle":"2023-05-22T10:13:56.994109Z","shell.execute_reply.started":"2023-05-22T10:13:56.980782Z","shell.execute_reply":"2023-05-22T10:13:56.992934Z"}}},{"cell_type":"code","source":"import json\nimport pickle\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Process","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/asl-fingerspelling/train.csv\")\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T13:40:28.123799Z","iopub.execute_input":"2023-05-29T13:40:28.124186Z","iopub.status.idle":"2023-05-29T13:40:28.314203Z","shell.execute_reply.started":"2023-05-29T13:40:28.124151Z","shell.execute_reply":"2023-05-29T13:40:28.313029Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Character Index","metadata":{}},{"cell_type":"code","source":"char_index = json.load(open(\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\"))\nprint(char_index)\n\nindex_char = dict([val, key] for key, val in char_index.items())\nprint(index_char)\n\nwith open('index.json','w') as f:\n    b = json.dump(index_char, f)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sequence List","metadata":{}},{"cell_type":"code","source":"sequence_id_list = train_df.sequence_id.tolist()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Label List","metadata":{}},{"cell_type":"code","source":"phrase_list = train_df.phrase.to_list()\nlabel_list = []\nfor p in phrase_list:\n    p = list(p)\n    label = []\n    for i in p:\n        label.append(char_index[i])\n    label_list.append(label)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Start List & Length List","metadata":{}},{"cell_type":"code","source":"dataset_path = \"/kaggle/input/asl-fingerspelling\"\ncur_file = ''\nlen_list = []\nstart_list = []\nlength = 0","metadata":{"execution":{"iopub.status.busy":"2023-05-29T13:40:28.466843Z","iopub.execute_input":"2023-05-29T13:40:28.467275Z","iopub.status.idle":"2023-05-29T13:40:28.48042Z","shell.execute_reply.started":"2023-05-29T13:40:28.467241Z","shell.execute_reply":"2023-05-29T13:40:28.478819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(range(len(train_df))):\n    file_name = train_df.path.iloc[i]\n    sequence_id = train_df.sequence_id.iloc[i]\n    if cur_file != file_name:\n        file_df = pd.read_parquet(f\"{dataset_path}/{file_name}\")\n        cur_file = file_name\n        file_df = file_df.reset_index()\n    \n    frames_df = file_df[file_df.sequence_id == sequence_id]\n    len_list.append(len(frames_df))\n    start_list.append(length)\n    length += len(frames_df)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T13:40:28.482451Z","iopub.execute_input":"2023-05-29T13:40:28.483645Z","iopub.status.idle":"2023-05-29T14:00:17.868286Z","shell.execute_reply.started":"2023-05-29T13:40:28.48359Z","shell.execute_reply":"2023-05-29T14:00:17.865122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LENGTH = 10749578\nLANDMARKS = 543\nHAND_INDEX = 468\nLANDMARK_LENGTH = 543 - 468","metadata":{"execution":{"iopub.status.busy":"2023-05-29T14:00:17.8728Z","iopub.execute_input":"2023-05-29T14:00:17.873348Z","iopub.status.idle":"2023-05-29T14:00:17.882853Z","shell.execute_reply.started":"2023-05-29T14:00:17.873303Z","shell.execute_reply":"2023-05-29T14:00:17.881904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicts = {'label_list':label_list,'sequence_id_list':sequence_id_list,'start_list':start_list,'length_list':len_list}\nwith open('train.pickle','wb') as f:\n    pickle.dump(dicts,f)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_npy = np.zeros((length,75,3))\n# train_npy = np.random.rand(LENGTH,75,3)\ntrain_npy.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-29T14:00:17.886013Z","iopub.execute_input":"2023-05-29T14:00:17.887016Z","iopub.status.idle":"2023-05-29T14:00:17.906037Z","shell.execute_reply.started":"2023-05-29T14:00:17.886972Z","shell.execute_reply":"2023-05-29T14:00:17.904288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(range(len(train_df))):\n    file_name = train_df.path.iloc[i]\n    sequence_id = train_df.sequence_id.iloc[i]\n    if cur_file != file_name:\n        file_df = pd.read_parquet(f\"{dataset_path}/{file_name}\")\n        cur_file = file_name\n        file_df = file_df.reset_index()\n\n    frames_df = file_df[file_df.sequence_id == sequence_id]\n    frames_df = frames_df.iloc[:, 2:]\n    frames = np.array(frames_df)\n    train_npy[start_list[i]:start_list[i]+len_list[i],:,0] = frames[:,LANDMARKS*0+HAND_INDEX:LANDMARKS*0+HAND_INDEX+LANDMARK_LENGTH]\n    train_npy[start_list[i]:start_list[i]+len_list[i],:,1] = frames[:,LANDMARKS*1+HAND_INDEX:LANDMARKS*1+HAND_INDEX+LANDMARK_LENGTH]\n    train_npy[start_list[i]:start_list[i]+len_list[i],:,2] = frames[:,LANDMARKS*2+HAND_INDEX:LANDMARKS*2+HAND_INDEX+LANDMARK_LENGTH]\n    \n# train_npy = np.array(train_list)\nprint(train_npy.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T14:00:17.90767Z","iopub.execute_input":"2023-05-29T14:00:17.908698Z","iopub.status.idle":"2023-05-29T14:21:42.318347Z","shell.execute_reply.started":"2023-05-29T14:00:17.908653Z","shell.execute_reply":"2023-05-29T14:21:42.316473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save('train.npy', train_npy)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T14:21:42.320317Z","iopub.execute_input":"2023-05-29T14:21:42.320764Z","iopub.status.idle":"2023-05-29T14:22:57.008056Z","shell.execute_reply.started":"2023-05-29T14:21:42.320723Z","shell.execute_reply":"2023-05-29T14:22:57.006654Z"},"trusted":true},"execution_count":null,"outputs":[]}]}