{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Data Description\n**Files**\n- `sup.npy`: supplemental data, a `(F,75,3)` shape numpy object array\n    - `` stand for the amount of total frames in supplemental dataset\n    - `75` is the amount of skeleton points(without face points)\n    - `3` is the `[x,y,z]` coordinates.\n- `sup.pickle`: supplemental labels, a dict\n    1. `label_list`: a list of labels, each label is a array of character's index\n    2. `sequence_id_list`: a list of sequence_id.\n    3. `start_list`: a list of each sequence index frames\n    4. `length_list`: a list of each sequence length","metadata":{}},{"cell_type":"markdown","source":"# Config","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-22T10:13:56.980393Z","iopub.execute_input":"2023-05-22T10:13:56.980816Z","iopub.status.idle":"2023-05-22T10:13:56.994109Z","shell.execute_reply.started":"2023-05-22T10:13:56.980782Z","shell.execute_reply":"2023-05-22T10:13:56.992934Z"}}},{"cell_type":"code","source":"import json\nimport pickle\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-05-30T07:29:43.212128Z","iopub.execute_input":"2023-05-30T07:29:43.212597Z","iopub.status.idle":"2023-05-30T07:29:43.220715Z","shell.execute_reply.started":"2023-05-30T07:29:43.21256Z","shell.execute_reply":"2023-05-30T07:29:43.219551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sup_df = pd.read_csv(\"/kaggle/input/asl-fingerspelling/supplemental_metadata.csv\")\nsup_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-30T07:29:44.984705Z","iopub.execute_input":"2023-05-30T07:29:44.98528Z","iopub.status.idle":"2023-05-30T07:29:45.139295Z","shell.execute_reply.started":"2023-05-30T07:29:44.985219Z","shell.execute_reply":"2023-05-30T07:29:45.138331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Character Index","metadata":{}},{"cell_type":"code","source":"char_index = json.load(open(\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\"))\nprint(char_index)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T07:29:48.648813Z","iopub.execute_input":"2023-05-30T07:29:48.649222Z","iopub.status.idle":"2023-05-30T07:29:48.658509Z","shell.execute_reply.started":"2023-05-30T07:29:48.64919Z","shell.execute_reply":"2023-05-30T07:29:48.656829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequence_id_list = sup_df.sequence_id.tolist()","metadata":{"execution":{"iopub.status.busy":"2023-05-30T07:31:45.09248Z","iopub.execute_input":"2023-05-30T07:31:45.092876Z","iopub.status.idle":"2023-05-30T07:31:45.09929Z","shell.execute_reply.started":"2023-05-30T07:31:45.092847Z","shell.execute_reply":"2023-05-30T07:31:45.09823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"phrase_list = sup_df.phrase.to_list()\nlabel_list = []\nfor p in phrase_list:\n    p = list(p)\n    label = []\n    for i in p:\n        label.append(char_index[i])\n    label_list.append(label)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T07:43:08.428073Z","iopub.execute_input":"2023-05-30T07:43:08.428498Z","iopub.status.idle":"2023-05-30T07:43:08.930375Z","shell.execute_reply.started":"2023-05-30T07:43:08.428465Z","shell.execute_reply":"2023-05-30T07:43:08.929337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Numpy Data","metadata":{}},{"cell_type":"markdown","source":"## Supplemental Data","metadata":{}},{"cell_type":"code","source":"dataset_path = \"/kaggle/input/asl-fingerspelling\"\ncur_file = ''\nlen_list = []\nstart_list = []\nlength = 0","metadata":{"execution":{"iopub.status.busy":"2023-05-29T13:40:28.466843Z","iopub.execute_input":"2023-05-29T13:40:28.467275Z","iopub.status.idle":"2023-05-29T13:40:28.48042Z","shell.execute_reply.started":"2023-05-29T13:40:28.467241Z","shell.execute_reply":"2023-05-29T13:40:28.478819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(range(len(sup_df))):\n    file_name = sup_df.path.iloc[i]\n    sequence_id = sup_df.sequence_id.iloc[i]\n    if cur_file != file_name:\n        file_df = pd.read_parquet(f\"{dataset_path}/{file_name}\")\n        cur_file = file_name\n        file_df = file_df.reset_index()\n    # iterate througe each frame and divide them by sequence_id\n#     sequence_list = file_df.sequence_id.unique()\n    frames_df = file_df[file_df.sequence_id == sequence_id]\n    len_list.append(len(frames_df))\n    start_list.append(length)\n    length += len(frames_df)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T13:40:28.482451Z","iopub.execute_input":"2023-05-29T13:40:28.483645Z","iopub.status.idle":"2023-05-29T14:00:17.868286Z","shell.execute_reply.started":"2023-05-29T13:40:28.48359Z","shell.execute_reply":"2023-05-29T14:00:17.865122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicts = {'label_list':label_list,'sequence_id_list':sequence_id_list,'start_list':start_list,'length_list':len_list}\nwith open('sup.pickle','wb') as f:\n    pickle.dump(dicts,f)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LENGTH = 10749578\nLANDMARKS = 543\nHAND_INDEX = 468\nLANDMARK_LENGTH = 543 - 468\n# length = int(LENGTH/7)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T14:00:17.8728Z","iopub.execute_input":"2023-05-29T14:00:17.873348Z","iopub.status.idle":"2023-05-29T14:00:17.882853Z","shell.execute_reply.started":"2023-05-29T14:00:17.873303Z","shell.execute_reply":"2023-05-29T14:00:17.881904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sup_npy = np.zeros((length,75,3))\n# train_npy = np.random.rand(LENGTH,75,3)\nsup_npy.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-29T14:00:17.886013Z","iopub.execute_input":"2023-05-29T14:00:17.887016Z","iopub.status.idle":"2023-05-29T14:00:17.906037Z","shell.execute_reply.started":"2023-05-29T14:00:17.886972Z","shell.execute_reply":"2023-05-29T14:00:17.904288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(range(len(sup_df))):\n    file_name = sup_df.path.iloc[i]\n    sequence_id = sup_df.sequence_id.iloc[i]\n    if cur_file != file_name:\n        file_df = pd.read_parquet(f\"{dataset_path}/{file_name}\")\n        cur_file = file_name\n        file_df = file_df.reset_index()\n\n    frames_df = file_df[file_df.sequence_id == sequence_id]\n    frames_df = frames_df.iloc[:, 2:]\n    frames = np.array(frames_df)\n    sup_npy[start_list[i]:start_list[i]+len_list[i],:,0] = frames[:,LANDMARKS*0+HAND_INDEX:LANDMARKS*0+HAND_INDEX+LANDMARK_LENGTH]\n    sup_npy[start_list[i]:start_list[i]+len_list[i],:,1] = frames[:,LANDMARKS*1+HAND_INDEX:LANDMARKS*1+HAND_INDEX+LANDMARK_LENGTH]\n    sup_npy[start_list[i]:start_list[i]+len_list[i],:,2] = frames[:,LANDMARKS*2+HAND_INDEX:LANDMARKS*2+HAND_INDEX+LANDMARK_LENGTH]\n    \n# train_npy = np.array(train_list)\nprint(sup_npy.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T14:00:17.90767Z","iopub.execute_input":"2023-05-29T14:00:17.908698Z","iopub.status.idle":"2023-05-29T14:21:42.318347Z","shell.execute_reply.started":"2023-05-29T14:00:17.908653Z","shell.execute_reply":"2023-05-29T14:21:42.316473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save('sup.npy', sup_npy)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T14:21:42.320317Z","iopub.execute_input":"2023-05-29T14:21:42.320764Z","iopub.status.idle":"2023-05-29T14:22:57.008056Z","shell.execute_reply.started":"2023-05-29T14:21:42.320723Z","shell.execute_reply":"2023-05-29T14:22:57.006654Z"},"trusted":true},"execution_count":null,"outputs":[]}]}