{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom skimage.transform import resize\nimport json\nfrom tqdm import tqdm\nimport os\ninpdir = \"/kaggle/input/asl-fingerspelling\"\n\ndf = pd.read_csv(f'{inpdir}/train.csv')\n\ndf[\"phrase_bytes\"] = df[\"phrase\"].map(lambda x: x.encode(\"utf-8\"))\ndisplay(df.head())\nLIP = [\n    61, 185, 40, 39, 37, 267, 269, 270, 409,\n    291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n    78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n    95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n]\n\nFACE = [f'x_face_{i}' for i in LIP] + [f'y_face_{i}' for i in LIP] + [f'z_face_{i}' for i in LIP]\nLHAND = [f'x_left_hand_{i}' for i in range(21)] + [f'y_left_hand_{i}' for i in range(21)] + [f'z_left_hand_{i}' for i in range(21)]\nRHAND = [f'x_right_hand_{i}' for i in range(21)] + [f'y_right_hand_{i}' for i in range(21)] + [f'z_right_hand_{i}' for i in range(21)]\nPOSE = [f'x_pose_{i}' for i in range(33)] + [f'y_pose_{i}' for i in range(33)] + [f'z_pose_{i}' for i in range(33)]\n\nSEL_COLS = FACE + LHAND + RHAND + POSE\nFRAME_LEN = 128\ndef load_relevant_data_subset(pq_path):\n    return pd.read_parquet(pq_path, columns=SEL_COLS)\n\nfor file_id in tqdm(df.file_id.unique()):\n    pqfile = f\"{inpdir}/train_landmarks/{file_id}.parquet\"\n    if not os.path.isdir(\"tfds\"): os.mkdir(\"tfds\")\n    tffile = f\"tfds/{file_id}.tfrecord\"\n    seq_refs = df.loc[df.file_id == file_id]\n    seqs = load_relevant_data_subset(pqfile)\n    \n    with tf.io.TFRecordWriter(tffile) as file_writer:\n        for seq_id, phrase in zip(seq_refs.sequence_id, seq_refs.phrase_bytes):\n            frames = seqs.iloc[seqs.index == seq_id]\n            frames128 = frames.fillna(-10).to_numpy()\n            frames128 = resize(frames128, (FRAME_LEN, len(SEL_COLS)))\n            frames = pd.DataFrame(data = frames128, columns=frames.columns)\n            #转成tfcord格式\n            features = {COL: tf.train.Feature(float_list=tf.train.FloatList(value=frames[COL])) for COL in SEL_COLS}\n            features[\"phrase\"] = tf.train.Feature(bytes_list=tf.train.BytesList(value=[phrase]))\n            record_bytes = tf.train.Example(features=tf.train.Features(feature=features)).SerializeToString()\n            file_writer.write(record_bytes)\ndef decode_fn(record_bytes):\n    schema = {COL: tf.io.FixedLenFeature([FRAME_LEN], dtype=tf.float32) for COL in SEL_COLS}\n    schema[\"phrase\"] = tf.io.FixedLenFeature([], dtype=tf.string)\n    return tf.io.parse_single_example(record_bytes, schema)\n\nfor file_id in df.file_id:\n    pqfile = f\"{inpdir}/train_landmarks/{file_id}.parquet\"\n    if not os.path.isdir(\"tfds\"): os.mkdir(\"tfds\")\n    tffile = f\"tfds/{file_id}.tfrecord\"\n    for batch in tf.data.TFRecordDataset([tffile]).map(decode_fn).take(2):\n        print(list(batch.keys())[0])\n    break","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-06T17:15:53.582809Z","iopub.execute_input":"2023-06-06T17:15:53.583192Z","iopub.status.idle":"2023-06-06T17:15:59.517461Z","shell.execute_reply.started":"2023-06-06T17:15:53.583155Z","shell.execute_reply":"2023-06-06T17:15:59.515863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom skimage.transform import resize\nimport json\nfrom tqdm import tqdm\nimport os\ninpdir = \"/kaggle/input/asl-fingerspelling\"\ndf = pd.read_csv(f'{inpdir}/train.csv')\nprint(df['phrase'])\ndf[\"phrase_bytes\"] = df[\"phrase\"].map(lambda x: x.encode(\"utf-8\"))\nprint(df['phrase_bytes'])\ndisplay(df.head(100))\nLIP = [\n    61, 185, 40, 39, 37, 267, 269, 270, 409,\n    291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n    78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n    95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n]\n\nFACE = [f'x_face_{i}' for i in LIP] + [f'y_face_{i}' for i in LIP] + [f'z_face_{i}' for i in LIP]\nLHAND = [f'x_left_hand_{i}' for i in range(21)] + [f'y_left_hand_{i}' for i in range(21)] + [f'z_left_hand_{i}' for i in range(21)]\nRHAND = [f'x_right_hand_{i}' for i in range(21)] + [f'y_right_hand_{i}' for i in range(21)] + [f'z_right_hand_{i}' for i in range(21)]\nPOSE = [f'x_pose_{i}' for i in range(33)] + [f'y_pose_{i}' for i in range(33)] + [f'z_pose_{i}' for i in range(33)]\n\nSEL_COLS = FACE + LHAND + RHAND + POSE\nFRAME_LEN = 128","metadata":{"execution":{"iopub.status.busy":"2023-06-07T06:56:00.786968Z","iopub.execute_input":"2023-06-07T06:56:00.78736Z","iopub.status.idle":"2023-06-07T06:56:00.889627Z","shell.execute_reply.started":"2023-06-07T06:56:00.787333Z","shell.execute_reply":"2023-06-07T06:56:00.888331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_relevant_data_subset(pq_path):\n    return pd.read_parquet(pq_path, columns=SEL_COLS)\nprint(pd.read_parquet('/kaggle/input/asl-fingerspelling/train_landmarks/1021040628.parquet',columns=SEL_COLS))\na = pd.read_parquet('/kaggle/input/asl-fingerspelling/train_landmarks/1021040628.parquet',columns=SEL_COLS)\n       ","metadata":{"execution":{"iopub.status.busy":"2023-06-07T06:47:14.002585Z","iopub.execute_input":"2023-06-07T06:47:14.003339Z","iopub.status.idle":"2023-06-07T06:47:18.301701Z","shell.execute_reply.started":"2023-06-07T06:47:14.003301Z","shell.execute_reply":"2023-06-07T06:47:18.299692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for file_id in tqdm(df.file_id.unique()):\n    pqfile = f\"{inpdir}/train_landmarks/{file_id}.parquet\"\n    if not os.path.isdir(\"tfds\"): os.mkdir(\"tfds\")\n    tffile = f\"tfds/{file_id}.tfrecord\"\n    seq_refs = df.loc[df.file_id == file_id]\n    print(seq_refs)\n    seqs = load_relevant_data_subset(pqfile)\n    print(seqs)\n#     print(seq_refs.sequence_id, '\\n',seq_refs.phrase_bytes)\n    for seq_id ,phrase in zip(seq_refs.sequence_id,seq_refs.phrase_bytes):\n            frames = seqs.iloc[seqs.index == seq_id]\n#             print(frames)\n            frames128 = frames.fillna(0).to_numpy()\n#             print(frames128,frames128.shape)\n            frames128 = resize(frames128, (FRAME_LEN, len(SEL_COLS)))\n#             print(frames128,frames128.shape)\n            frames = pd.DataFrame(data = frames128,columns = frames.columns)\n            print(frames)\n            \n            \n            ","metadata":{"execution":{"iopub.status.busy":"2023-06-07T07:06:57.412082Z","iopub.execute_input":"2023-06-07T07:06:57.41252Z","iopub.status.idle":"2023-06-07T07:07:02.106878Z","shell.execute_reply.started":"2023-06-07T07:06:57.412485Z","shell.execute_reply":"2023-06-07T07:07:02.104968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}