{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from tqdm import tqdm\nimport multiprocessing as mp\nimport pandas as pd\nimport numpy as np\nimport os\nimport shutil\nimport argparse\nimport json","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class args:\n    input_dir='/kaggle/input/asl-fingerspelling/train_landmarks'\n    output_dir='./train_landmarks_npy/'\n    n_cores=4\n    train_df='/kaggle/input/asl-fingerspelling/train.csv'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(args.train_df)\n\n# train_cols in right order\nall_cols = [f'face_{i}' for i in range(468)] \nall_cols += [f'left_hand_{i}' for i in range(21)] \nall_cols += [f'pose_{i}' for i in range(33)]\nall_cols += [f'right_hand_{i}' for i in range(21)]\nall_cols = np.array(all_cols)\n\n\n#1st place kept landmarks\n\nNOSE=[\n    1,2,98,327\n]\nLNOSE = [98]\nRNOSE = [327]\nLIP = [ 0, \n    61, 185, 40, 39, 37, 267, 269, 270, 409,\n    291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n    78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n    95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n]\nLLIP = [84,181,91,146,61,185,40,39,37,87,178,88,95,78,191,80,81,82]\nRLIP = [314,405,321,375,291,409,270,269,267,317,402,318,324,308,415,310,311,312]\n\nPOSE = [500, 502, 504, 501, 503, 505, 512, 513]\nLPOSE = [513,505,503,501]\nRPOSE = [512,504,502,500]\n\nLARMS = [501, 503, 505, 507, 509, 511]\nRARMS = [500, 502, 504, 506, 508, 510]\n\nREYE = [\n    33, 7, 163, 144, 145, 153, 154, 155, 133,\n    246, 161, 160, 159, 158, 157, 173,\n]\nLEYE = [\n    263, 249, 390, 373, 374, 380, 381, 382, 362,\n    466, 388, 387, 386, 385, 384, 398,\n]\n\nLHAND = np.arange(468, 489).tolist()\nRHAND = np.arange(522, 543).tolist()\n\nPOINT_LANDMARKS = LIP + LHAND + RHAND + NOSE + REYE + LEYE + LARMS + RARMS\n\nkept_cols = all_cols[POINT_LANDMARKS]\nn_landmarks = len(kept_cols)\n\nkept_cols_xyz = np.array(['x_' + c for c in kept_cols] + ['y_' + c for c in kept_cols] + ['z_' + c for c in kept_cols])\n\n\nTARGET_FOLDER = args.output_dir\n\nfile_ids = train['file_id'].unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def do_one(file_id):\n    os.makedirs(TARGET_FOLDER + f'{file_id}/', exist_ok=True)\n    df = pd.read_parquet(f'{args.input_dir}/{file_id}.parquet').reset_index()\n    sequence_ids = df['sequence_id'].unique()\n    for sequence_id in sequence_ids:\n        df_seq = df[df['sequence_id']==sequence_id].copy()\n        vals = df_seq[kept_cols_xyz].values\n        np.save(TARGET_FOLDER + f'{file_id}/{sequence_id}.npy',vals)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not os.path.exists(args.output_dir):\n    os.makedirs(args.output_dir)\n#shutil.copy(args.train_df, args.output_dir + '../')\n#shutil.copy('/kaggle/input/asl-fingerspelling/character_to_prediction_index.json', args.output_dir + '../')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# multiprocessing.freeze_support()\nwith mp.Pool(args.n_cores) as p:\n    res = list(tqdm(p.imap(do_one,file_ids), total=len(file_ids)))\n\nselected_columns_dict = {\"selected_columns\": kept_cols_xyz.tolist()}\nwith open(f'{TARGET_FOLDER}inference_args.json', \"w\") as f:\n    json.dump(selected_columns_dict, f)\n\nnp.save(TARGET_FOLDER + 'columns.npy',kept_cols_xyz)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}