{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook is for processing supplemental dataset. (Reference: https://www.kaggle.com/code/irohith/aslfr-preprocess-dataset)","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom skimage.transform import resize\nimport json\nfrom tqdm import tqdm\nimport os\nimport shutil\nfrom tqdm import tqdm\nimport tensorflow as tf","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT = \"/kaggle/input/asl-fingerspelling\"\nOUTPUT_ROOT = \"/kaggle/working/\"","metadata":{"execution":{"iopub.status.busy":"2023-06-11T02:46:59.16203Z","iopub.execute_input":"2023-06-11T02:46:59.162419Z","iopub.status.idle":"2023-06-11T02:46:59.167135Z","shell.execute_reply.started":"2023-06-11T02:46:59.162386Z","shell.execute_reply":"2023-06-11T02:46:59.166196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# supplemental_landmarks\n\ndf = pd.read_csv(f'{ROOT}/supplemental_metadata.csv')\ndf[\"phrase_bytes\"] = df[\"phrase\"].map(lambda x: x.encode(\"utf-8\"))\ndisplay(df.head())","metadata":{"execution":{"iopub.status.busy":"2023-06-11T02:34:35.432846Z","iopub.execute_input":"2023-06-11T02:34:35.433282Z","iopub.status.idle":"2023-06-11T02:34:35.611253Z","shell.execute_reply.started":"2023-06-11T02:34:35.433249Z","shell.execute_reply":"2023-06-11T02:34:35.610179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You can find coordinates from [mediapipe github](https://github.com/google/mediapipe/issues/1049#issuecomment-707847994).","metadata":{}},{"cell_type":"code","source":"LIP = [\n    61, 185, 40, 39, 37, 267, 269, 270, 409,\n    291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n    78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n    95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n]\n\nFACE = [f'x_face_{i}' for i in LIP] + [f'y_face_{i}' for i in LIP] + [f'z_face_{i}' for i in LIP]\nLHAND = [f'x_left_hand_{i}' for i in range(21)] + [f'y_left_hand_{i}' for i in range(21)] + [f'z_left_hand_{i}' for i in range(21)]\nRHAND = [f'x_right_hand_{i}' for i in range(21)] + [f'y_right_hand_{i}' for i in range(21)] + [f'z_right_hand_{i}' for i in range(21)]\nPOSE = [f'x_pose_{i}' for i in range(33)] + [f'y_pose_{i}' for i in range(33)] + [f'z_pose_{i}' for i in range(33)]\n\nSEL_COLS = FACE + LHAND + RHAND + POSE\nFRAME_LEN = 128\n\nprint(SEL_COLS[:5])","metadata":{"execution":{"iopub.status.busy":"2023-06-11T02:45:11.850018Z","iopub.execute_input":"2023-06-11T02:45:11.850472Z","iopub.status.idle":"2023-06-11T02:45:11.861031Z","shell.execute_reply.started":"2023-06-11T02:45:11.850438Z","shell.execute_reply":"2023-06-11T02:45:11.860038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A new column named \"phrase_bytes\" is added to the DataFrame `df` using the `.map()` function. This column is created by applying a lambda function to the \"phrase\" column of `df`. The lambda function encodes each string value in the \"phrase\" column using the UTF-8 encoding, converting it into a sequence of bytes. \n\nThe purpose of adding the `phrase_bytes` column is to store the phrase information in a format that can be serialized and written to a TFRecord file later in the code.","metadata":{}},{"cell_type":"code","source":"inpdir = \"/kaggle/input/asl-fingerspelling\"\ndf = pd.read_csv(f'{inpdir}/supplemental_metadata.csv')\ndf[\"phrase_bytes\"] = df[\"phrase\"].map(lambda x: x.encode(\"utf-8\"))\ndisplay(df.head())","metadata":{"execution":{"iopub.status.busy":"2023-06-11T02:53:09.154347Z","iopub.execute_input":"2023-06-11T02:53:09.154791Z","iopub.status.idle":"2023-06-11T02:53:09.253895Z","shell.execute_reply.started":"2023-06-11T02:53:09.154732Z","shell.execute_reply":"2023-06-11T02:53:09.252713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This function loads a subset of relevant columns from a Parquet file specified by pq_path and returns it as a DataFrame. And it checks if the directory named \"tfds_supp\" doesn't exist in the OUTPUT_ROOT directory, and if so, creates the \"tfds_supp\" directory.","metadata":{}},{"cell_type":"code","source":"def load_relevant_data_subset(pq_path):\n    return pd.read_parquet(pq_path, columns=SEL_COLS)\n\n# Create directory \"tfds_supp\" if it doesn't exist\nif not os.path.isdir(f\"{OUTPUT_ROOT}/tfds_supp\"):\n    os.mkdir(f\"{OUTPUT_ROOT}/tfds_supp\")","metadata":{"execution":{"iopub.status.busy":"2023-06-11T02:53:12.328302Z","iopub.execute_input":"2023-06-11T02:53:12.328684Z","iopub.status.idle":"2023-06-11T02:53:12.335125Z","shell.execute_reply.started":"2023-06-11T02:53:12.328655Z","shell.execute_reply":"2023-06-11T02:53:12.333718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This loop iterates over unique file_id values in the df DataFrame. For each file_id, it constructs the paths for the Parquet file (pqfile) and the TFRecord file (tffile). It retrieves the sequence references (seq_refs) from the DataFrame for the current file_id, and then loads the relevant data subset (`seq","metadata":{}},{"cell_type":"code","source":"# Iterate over unique file_ids in the DataFrame\nfor file_id in tqdm(df.file_id.unique()):\n    pqfile = f'{ROOT}/supplemental_landmarks/{file_id}.parquet'  # Path to the Parquet file\n    tffile = f\"{OUTPUT_ROOT}/tfds_supp/{file_id}.tfrecord\"  # Path to the TFRecord file\n    seq_refs = df.loc[df.file_id == file_id]  # Get sequence references for the current file_id\n    seqs = load_relevant_data_subset(pqfile)  # Load relevant data subset from the Parquet file\n\n    # Open a TFRecordWriter to write the TFRecord file\n    with tf.io.TFRecordWriter(tffile) as file_writer:\n        # Iterate over sequence_id and phrase for each sequence reference\n        for seq_id, phrase in zip(seq_refs.sequence_id, seq_refs.phrase_bytes):\n            frames = seqs.iloc[seqs.index == seq_id]  # Get the frames for the current sequence_id\n            frames128 = frames.fillna(-10).to_numpy()  # Replace missing values and convert frames to a NumPy array\n\n            if frames128.size == 0:  # Skip if frames128 is empty\n                continue\n\n            frames128 = resize(frames128, (FRAME_LEN, len(SEL_COLS)))  # Resize the frames to a fixed length\n            frames = pd.DataFrame(data=frames128, columns=frames.columns)  # Convert the resized frames back to a DataFrame\n\n            # Create features for the TFRecord\n            features = {COL: tf.train.Feature(float_list=tf.train.FloatList(value=frames[COL])) for COL in SEL_COLS}\n            features[\"phrase\"] = tf.train.Feature(bytes_list=tf.train.BytesList(value=[phrase]))\n\n            # Serialize the features into a TFRecord example and write it to the file\n            record_bytes = tf.train.Example(features=tf.train.Features(feature=features)).SerializeToString()\n            file_writer.write(record_bytes)","metadata":{"execution":{"iopub.status.busy":"2023-06-11T02:53:15.591418Z","iopub.execute_input":"2023-06-11T02:53:15.591895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfds_supp_path = f\"{OUTPUT_ROOT}/tfds_supp\"  # Path to the \"tfds_supp\" directory\nmetadata_path = f\"{ROOT}/supplemental_metadata.csv\"  # Path to the \"supplemental_metadata.csv\" file\n\n# Check the number of files in the \"tfds_supp\" directory\ntfds_supp_files = os.listdir(tfds_supp_path)\nnum_files_in_tfds_supp = len(tfds_supp_files)\n\n# Check the number of unique paths in the \"supplemental_metadata.csv\" file\nmetadata_data = pd.read_csv(metadata_path)\nnum_unique_paths_in_metadata = metadata_data['path'].nunique()\n\n# Print the results\nprint(f\"Number of files in tfds_supp: {num_files_in_tfds_supp}\")\nprint(f\"Number of unique paths in supplemental_metadata.csv: {num_unique_paths_in_metadata}\")\n\n# Check if the number of files and unique paths are the same\nif num_files_in_tfds_supp == num_unique_paths_in_metadata:\n    print(\"Number of files in tfds_supp and number of unique paths in supplemental_metadata.csv are the same.\")\nelse:\n    print(\"Number of files in tfds_supp and number of unique paths in supplemental_metadata.csv are different.\")\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}