{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook converts the data into TfRecord files. Currently it reduces the alnd_mark features to only the hand x/y coordinate(x_y_hand_features) However, you could modify this pretty easily if you wanted to apply some other preprocessing prior to creating the TfRecord, such as only saving some of the features.\n\nPlease comment any ideas to improve this.\n\nThank you and Good Luck!","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport pandas as pd\nimport json\nimport os\nfrom tqdm import tqdm\nimport shutil","metadata":{"execution":{"iopub.status.busy":"2023-05-26T16:28:41.858361Z","iopub.execute_input":"2023-05-26T16:28:41.85887Z","iopub.status.idle":"2023-05-26T16:28:41.866246Z","shell.execute_reply.started":"2023-05-26T16:28:41.858833Z","shell.execute_reply":"2023-05-26T16:28:41.864759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\ntrain","metadata":{"execution":{"iopub.status.busy":"2023-05-26T15:51:27.704357Z","iopub.execute_input":"2023-05-26T15:51:27.705311Z","iopub.status.idle":"2023-05-26T15:51:27.919871Z","shell.execute_reply.started":"2023-05-26T15:51:27.705266Z","shell.execute_reply":"2023-05-26T15:51:27.918559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def serialize_example(example):\n    '''\n    Function used write the tfrecord\n    coordinates_shape is used the reshape the the record when loading, as not all the samples have the same shape\n    coordinates is the values from the train_landmarks\n    phrase is from the train df, what we are trying to predict\n    '''\n    coordinates_shape = example['coordinates_shape']\n    coordinates_bytes = example['coordinates'].astype(np.float32).tobytes()\n    phrase_bytes = example['phrase'].encode('utf-8')\n\n    feature = {\n        'coordinates': tf.train.Feature(bytes_list=tf.train.BytesList(value=[coordinates_bytes])),\n        'coordinates_shape': tf.train.Feature(int64_list=tf.train.Int64List(value=coordinates_shape)),\n        'phrase': tf.train.Feature(bytes_list=tf.train.BytesList(value=[phrase_bytes])),\n    }\n\n    example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example_proto.SerializeToString()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-26T15:51:27.92132Z","iopub.execute_input":"2023-05-26T15:51:27.921738Z","iopub.status.idle":"2023-05-26T15:51:27.932826Z","shell.execute_reply.started":"2023-05-26T15:51:27.921706Z","shell.execute_reply":"2023-05-26T15:51:27.931318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_parquet('/kaggle/input/asl-fingerspelling/train_landmarks/1019715464.parquet')\ndf = df.reset_index()\ndf\n\nseq_df = df[df['sequence_id'] == 1975433633]\nx_y_hand_features = []\n\nfor count, col in enumerate(seq_df.columns):\n    if 'hand' in col:\n        if 'z' not in col:\n            x_y_hand_features.append(count)\n    \nprint(x_y_hand_features)\n\nseq_df2 = seq_df.iloc[:, x_y_hand_features]\nseq_df2","metadata":{"execution":{"iopub.status.busy":"2023-05-26T16:02:11.827657Z","iopub.execute_input":"2023-05-26T16:02:11.828234Z","iopub.status.idle":"2023-05-26T16:02:14.91735Z","shell.execute_reply.started":"2023-05-26T16:02:11.828193Z","shell.execute_reply":"2023-05-26T16:02:14.916161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_data(df):\n    '''\n    Takes in a list of sequnce ids from each parquet file\n    Gets the landmark feature values and the phrase\n    Returns these as an example for the tfrecord creation\n    '''\n    data = []\n    sequence_ids = df['sequence_id'].unique()\n    for sequence_id in tqdm(sequence_ids):       \n        seq_df = df[df['sequence_id'] == sequence_id]\n        seq_df_reduced_features = seq_df.iloc[:, x_y_hand_features]\n        x = seq_df_reduced_features.values\n        x_shape = x.shape\n        #print(x_shape)\n        y = train.loc[train['sequence_id'] == sequence_id, 'phrase'].values[0]\n        example = {\n            'coordinates': x.astype(np.float32),\n            'coordinates_shape': list(x_shape),\n            'phrase': y\n        }\n        #print(\"Coordinates shape:\", example['coordinates'].shape)\n        data.append(example)\n    return data","metadata":{"execution":{"iopub.status.busy":"2023-05-26T16:03:53.406777Z","iopub.execute_input":"2023-05-26T16:03:53.407178Z","iopub.status.idle":"2023-05-26T16:03:53.415423Z","shell.execute_reply.started":"2023-05-26T16:03:53.407148Z","shell.execute_reply":"2023-05-26T16:03:53.414484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_tfrecord(parquet_path, TfRecordFileName):\n    '''\n    takes in a parquet file\n    Creates data, which is a list of examples  \n    TFRecordWriter iterates through each example data, creating a TfRecord file\n    '''\n    df = pd.read_parquet(parquet_path)\n    df = df.reset_index()\n    \n    #print(sequence_ids)\n    data = create_data(df)\n#     for example in data:\n#         coordinates_shape = example['coordinates_shape']\n#         print(\"Coordinates shape:\", coordinates_shape)\n    with tf.io.TFRecordWriter(TfRecordFileName) as writer:\n        for example in data:\n            example_proto = serialize_example(example)\n            writer.write(example_proto)\n        writer.close()\n    \n    \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2023-05-26T16:04:17.149095Z","iopub.execute_input":"2023-05-26T16:04:17.149567Z","iopub.status.idle":"2023-05-26T16:04:17.157709Z","shell.execute_reply.started":"2023-05-26T16:04:17.149534Z","shell.execute_reply":"2023-05-26T16:04:17.156381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_tfrecord('/kaggle/input/asl-fingerspelling/train_landmarks/1365772051.parquet', 'test2.tfrecord')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate over parquet files creating a tfrecord for each\nbase_path = '/kaggle/input/asl-fingerspelling/train_landmarks'\nfor count, parq in tqdm(enumerate(os.listdir(base_path))):\n    path = os.path.join(base_path, parq)\n    TfRecordFileName = f'TfRecord{count}.tfrecord'\n    create_tfrecord(path, TfRecordFileName)\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2023-05-26T16:04:45.746655Z","iopub.execute_input":"2023-05-26T16:04:45.747075Z","iopub.status.idle":"2023-05-26T16:27:05.740123Z","shell.execute_reply.started":"2023-05-26T16:04:45.747037Z","shell.execute_reply":"2023-05-26T16:27:05.737879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def parse_example(serialized_example):\n    feature_description = {\n        'coordinates': tf.io.FixedLenFeature([], tf.string),\n        'coordinates_shape': tf.io.FixedLenFeature([2], tf.int64),\n        'phrase': tf.io.FixedLenFeature([], tf.string)\n    }\n    example = tf.io.parse_single_example(serialized_example, feature_description)\n\n    # Decode the  array\n    coordinates = tf.io.decode_raw(example['coordinates'], tf.float32)\n    \n    # Reshape the array using the stored shape information\n    coordinates_shape = example['coordinates_shape']\n    coordinates = tf.reshape(coordinates, coordinates_shape)\n    \n    # Decode the label feature\n    phrase = tf.strings.strip(example['phrase'])\n\n    return {'coordinates': coordinates, 'phrase': phrase}","metadata":{"execution":{"iopub.status.busy":"2023-05-26T16:27:47.996993Z","iopub.execute_input":"2023-05-26T16:27:47.998339Z","iopub.status.idle":"2023-05-26T16:27:48.01036Z","shell.execute_reply.started":"2023-05-26T16:27:47.998283Z","shell.execute_reply":"2023-05-26T16:27:48.00862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example for test\ndf = pd.read_parquet('/kaggle/input/asl-fingerspelling/train_landmarks/1098899348.parquet')\ndf = df.reset_index()\ndf\n\n# Create Test TfRecord\ncreate_tfrecord('/kaggle/input/asl-fingerspelling/train_landmarks/1098899348.parquet', 'TfRecordTest.tfrecord')","metadata":{"execution":{"iopub.status.busy":"2023-05-26T16:27:48.838978Z","iopub.execute_input":"2023-05-26T16:27:48.839552Z","iopub.status.idle":"2023-05-26T16:28:12.949054Z","shell.execute_reply.started":"2023-05-26T16:27:48.839506Z","shell.execute_reply":"2023-05-26T16:28:12.947513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-05-26T16:28:12.951996Z","iopub.execute_input":"2023-05-26T16:28:12.952761Z","iopub.status.idle":"2023-05-26T16:28:13.019078Z","shell.execute_reply.started":"2023-05-26T16:28:12.952722Z","shell.execute_reply":"2023-05-26T16:28:13.017899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Test\n# TFRecord file path\nfilename = '/kaggle/working/TfRecord13.tfrecord'\n\n# Create a dataset from the TFRecord file\ndataset = tf.data.TFRecordDataset(filename)\n\n# Map the parse_example function to decode the examples\ndecoded_dataset = dataset.map(parse_example)\n\n# Iterate over the decoded dataset\nfor example in decoded_dataset:\n    print('coordinates:')\n    print(example['coordinates'].shape)\n    print('phrase:', example['phrase'])\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T16:28:13.020783Z","iopub.execute_input":"2023-05-26T16:28:13.021266Z","iopub.status.idle":"2023-05-26T16:28:13.853549Z","shell.execute_reply.started":"2023-05-26T16:28:13.021217Z","shell.execute_reply":"2023-05-26T16:28:13.852203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}