{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\n\ndataset = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\nprint(\"Full train dataset shape is {}\".format(dataset.shape))\n\n# Explore the data\nprint(dataset.head())","metadata":{"execution":{"iopub.status.busy":"2023-08-20T22:13:15.029084Z","iopub.execute_input":"2023-08-20T22:13:15.029848Z","iopub.status.idle":"2023-08-20T22:13:15.236917Z","shell.execute_reply.started":"2023-08-20T22:13:15.029711Z","shell.execute_reply":"2023-08-20T22:13:15.235639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The script defines and processes 3D landmarks for hand gestures and body poses, organizing them into coordinate labels.\n* It then reads gesture data from parquet files, filters sequences based on non-NaN values, and stores the preprocessed data in the TFRecord format in a designated directory.","metadata":{}},{"cell_type":"code","source":"# import Libraries\nimport os\nimport shutil\nimport numpy as np\nimport pyarrow.parquet as pq\nimport tensorflow as tf\nfrom tqdm import tqdm\n\n# Define pose coordinates\nLPOSE = [13, 15, 17, 19, 21]\nRPOSE = [14, 16, 18, 20, 22]\nPOSE = LPOSE + RPOSE\n\n# Create coordinate labels\nX = [f'x_right_hand_{i}' for i in range(21)] + [f'x_left_hand_{i}' for i in range(21)] + [f'x_pose_{i}' for i in POSE]\nY = [f'y_right_hand_{i}' for i in range(21)] + [f'y_left_hand_{i}' for i in range(21)] + [f'y_pose_{i}' for i in POSE]\nZ = [f'z_right_hand_{i}' for i in range(21)] + [f'z_left_hand_{i}' for i in range(21)] + [f'z_pose_{i}' for i in POSE]\n\n# Feature columns\nFEATURE_COLUMNS = X + Y + Z","metadata":{"execution":{"iopub.status.busy":"2023-08-20T22:13:15.241883Z","iopub.execute_input":"2023-08-20T22:13:15.242666Z","iopub.status.idle":"2023-08-20T22:13:24.908989Z","shell.execute_reply.started":"2023-08-20T22:13:15.24262Z","shell.execute_reply":"2023-08-20T22:13:24.907903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create index lists for easy access of landmarks\nX_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"x_\" in col]\nY_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"y_\" in col]\nZ_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"z_\" in col]\n\nRHAND_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"right\" in col]\nLHAND_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"left\" in col]\nRPOSE_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"pose\" in col and int(col[-2:]) in RPOSE]\nLPOSE_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"pose\" in col and int(col[-2:]) in LPOSE]","metadata":{"execution":{"iopub.status.busy":"2023-08-20T22:13:24.910779Z","iopub.execute_input":"2023-08-20T22:13:24.911576Z","iopub.status.idle":"2023-08-20T22:13:24.920434Z","shell.execute_reply.started":"2023-08-20T22:13:24.911517Z","shell.execute_reply":"2023-08-20T22:13:24.919148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create directory for preprocessed data\nif not os.path.isdir(\"preprocessed_tfrecords\"):\n    os.mkdir(\"preprocessed_tfrecords\")\nelse:\n    shutil.rmtree(\"preprocessed_tfrecords\")\n    os.mkdir(\"preprocessed_tfrecords\")\n\n# Process each file in the dataset\nfor file_id in tqdm(dataset.file_id.unique()):\n    # Extract file-specific information\n    file_df = dataset.loc[dataset[\"file_id\"] == file_id]\n    parquet_df = pq.read_table(f\"/kaggle/input/asl-fingerspelling/train_landmarks/{str(file_id)}.parquet\",\n                              columns=['sequence_id'] + FEATURE_COLUMNS).to_pandas()\n    parquet_numpy = parquet_df.to_numpy()\n    tf_file = f\"preprocessed_tfrecords/{file_id}.tfrecord\"\n    \n    # Open TFRecord file\n    with tf.io.TFRecordWriter(tf_file) as file_writer:\n        # Loop through each sequence in file.\n        for seq_id, phrase in zip(file_df.sequence_id, file_df.phrase):\n            # Fetch sequence data\n            frames = parquet_numpy[parquet_df.index == seq_id]\n            \n            # Calculate the number of NaN values in each hand landmark\n            r_nonan = np.sum(np.sum(np.isnan(frames[:, RHAND_IDX]), axis = 1) == 0)\n            l_nonan = np.sum(np.sum(np.isnan(frames[:, LHAND_IDX]), axis = 1) == 0)\n            no_nan = max(r_nonan, l_nonan)\n            \n            if 2*len(phrase)<no_nan:\n                features = {FEATURE_COLUMNS[i]: tf.train.Feature(\n                    float_list=tf.train.FloatList(value=frames[:, i])) for i in range(len(FEATURE_COLUMNS))}\n                features[\"phrase\"] = tf.train.Feature(bytes_list=tf.train.BytesList(value=[bytes(phrase, 'utf-8')]))\n                record_bytes = tf.train.Example(features=tf.train.Features(feature=features)).SerializeToString()\n                file_writer.write(record_bytes)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-20T22:13:24.924682Z","iopub.execute_input":"2023-08-20T22:13:24.925067Z","iopub.status.idle":"2023-08-20T22:24:32.144886Z","shell.execute_reply.started":"2023-08-20T22:13:24.925025Z","shell.execute_reply":"2023-08-20T22:24:32.14261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get list of all the TFRecord files\ntfrecords = dataset['file_id'].map(lambda x: f'/kaggle/working/preprocessed_tfrecords/{x}.tfrecord').unique()\n\n\nprint(f\"List of {len(tfrecords)} TFRecord files.\")","metadata":{"execution":{"iopub.status.busy":"2023-08-20T22:24:32.14668Z","iopub.execute_input":"2023-08-20T22:24:32.147245Z","iopub.status.idle":"2023-08-20T22:24:32.257397Z","shell.execute_reply.started":"2023-08-20T22:24:32.147203Z","shell.execute_reply":"2023-08-20T22:24:32.256407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# Check if all TFRecord files exist and are not empty\nexisting_tfrecords = [tfrecord for tfrecord in tfrecords if os.path.isfile(tfrecord) and os.path.getsize(tfrecord) > 0]\n\nprint(f\"Number of existing and non-empty TFRecord files: {len(existing_tfrecords)}\")\n\n# If there are files missing or empty, print their paths\nmissing_tfrecords = set(tfrecords) - set(existing_tfrecords)\nif missing_tfrecords:\n    print(\"The following TFRecord files are missing or empty:\")\n    for missing_tfrecord in missing_tfrecords:\n        print(missing_tfrecord)","metadata":{"execution":{"iopub.status.busy":"2023-08-20T22:24:32.262042Z","iopub.execute_input":"2023-08-20T22:24:32.265087Z","iopub.status.idle":"2023-08-20T22:24:32.278257Z","shell.execute_reply.started":"2023-08-20T22:24:32.265049Z","shell.execute_reply":"2023-08-20T22:24:32.277106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**JSON character mapping**\n\n\nThe script reads a JSON character mapping, then extends this mapping to include padding, start, and end tokens. It defines functions to preprocess gesture data by selecting the dominant hand (the one with fewer missing values) and then normalizes, pads, and reshapes the sequences for machine learning usage. The primary goal is to adjust and clean the data for further analysis or model training.","metadata":{}},{"cell_type":"code","source":"import json\nimport tensorflow as tf\n\n# Load and modify the json mapping\nwith open(\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\", \"r\") as json_file:\n    character_mapping = json.load(json_file)\n    \n# Add pad_token, start pointer and end pointer to the dict\npad_token = 'P'\nstart_token = '<'\nend_token = '>'\npad_token_idx = 59\nstart_token_idx = 60\nend_token_idx = 61\n\n# Add padding, start, and end tokens to the dictionary\ncharacter_mapping.update({\n    'P': 59,  # Padding token\n    '<': 60,  # Start token\n    '>': 61   # End token\n})\n\n# Create the reverse mapping\nnum_to_char = {value: key for key, value in character_mapping.items()}\n\nFRAME_LEN = 128\n\n# Function to adjust the shape and pad sequences\ndef adjust_and_pad(x):\n    x_shape = tf.shape(x)\n    if x_shape[0] < FRAME_LEN:\n        padding = [[0, FRAME_LEN - x_shape[0]], [0, 0], [0, 0]]\n        x = tf.pad(x, padding)\n    else:\n        new_shape = (FRAME_LEN, x_shape[1])\n        x = tf.image.resize(x, new_shape)\n    return x\n\n# Function to preprocess the data and detect the dominant hand\ndef preprocess_data(x):\n    right_hand_data = tf.gather(x, RHAND_IDX, axis=1)\n    left_hand_data = tf.gather(x, LHAND_IDX, axis=1)\n    right_pose_data = tf.gather(x, RPOSE_IDX, axis=1)\n    left_pose_data = tf.gather(x, LPOSE_IDX, axis=1)\n    \n    right_hand_nan_indices = tf.reduce_any(tf.math.is_nan(right_hand_data), axis=1)\n    left_hand_nan_indices = tf.reduce_any(tf.math.is_nan(left_hand_data), axis=1)\n    \n    right_hand_nans = tf.math.count_nonzero(right_hand_nan_indices)\n    left_hand_nans = tf.math.count_nonzero(left_hand_nan_indices)\n    \n    # Use the hand with less NaNs as the dominant hand\n    if right_hand_nans > left_hand_nans:\n        hand_data = left_hand_data\n        pose_data = left_pose_data\n    else:\n        hand_data = right_hand_data\n        pose_data = right_pose_data\n\n    # Normalize and concatenate hand and pose data\n    x = normalize_and_concatenate(hand_data, pose_data)\n\n    # Resize and pad sequence\n    x = adjust_and_pad(x)\n\n    # Replace NaNs with zeros\n    x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n\n    # Reshape to the desired format\n    x = tf.reshape(x, (FRAME_LEN, len(LHAND_IDX) + len(LPOSE_IDX)))\n\n    return x\n\ndef normalize_and_concatenate(hand, pose):\n    # Create coordinates for hand and pose\n    hand_x, hand_y, hand_z = split_coordinates(hand, LHAND_IDX)\n    pose_x, pose_y, pose_z = split_coordinates(pose, LPOSE_IDX)\n\n    # Normalize hand and pose coordinates\n    hand = normalize_coordinates(hand_x, hand_y, hand_z)\n    pose = normalize_coordinates(pose_x, pose_y, pose_z)\n\n    # Concatenate hand and pose data\n    return tf.concat([hand, pose], axis=1)\n\ndef split_coordinates(data, indices):\n    third = len(indices) // 3\n    x = data[:, 0*third : 1*third]\n    y = data[:, 1*third : 2*third]\n    z = data[:, 2*third : 3*third]\n    return x, y, z\n\ndef normalize_coordinates(x, y, z):\n    # Add new dimension to each coordinate for concatenation\n    x, y, z = [coordinate[..., tf.newaxis] for coordinate in (x, y, z)]\n    # Concatenate along the last dimension and normalize\n    data = tf.concat([x, y, z], axis=-1)\n    mean = tf.math.reduce_mean(data, axis=1, keepdims=True)\n    std = tf.math.reduce_std(data, axis=1, keepdims=True)\n    return (data - mean) / std","metadata":{"execution":{"iopub.status.busy":"2023-08-20T22:24:32.283772Z","iopub.execute_input":"2023-08-20T22:24:32.286159Z","iopub.status.idle":"2023-08-20T22:24:32.318344Z","shell.execute_reply.started":"2023-08-20T22:24:32.286122Z","shell.execute_reply":"2023-08-20T22:24:32.317004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**parse_record**\n\n\n* The parse_record function takes raw TFRecord bytes and converts them into a user-friendly format. It uses a defined schema to read landmark data and phrases from the TFRecord. Landmarks are transposed to maintain their original shape.\n\n* A lookup table is created using the character-to-number mapping. This table allows easy conversion of characters in a phrase to their corresponding numerical identifiers.\n\n* The transform_data function adds start and end tokens to a given phrase, converts characters to numbers using the lookup table, pads the phrase to ensure uniform length, and preprocesses the landmarks.","metadata":{}},{"cell_type":"code","source":"def parse_record(record_bytes):\n    # Create a dictionary with data schema for TFRecord\n    data_schema = {column: tf.io.VarLenFeature(dtype=tf.float32) for column in FEATURE_COLUMNS}\n    data_schema[\"phrase\"] = tf.io.FixedLenFeature([], dtype=tf.string)\n    \n    # Parse the record into our schema\n    parsed_features = tf.io.parse_single_example(record_bytes, data_schema)\n    \n    # Extract phrase from the parsed features\n    parsed_phrase = parsed_features[\"phrase\"]\n    \n    # Convert landmarks to dense tensor and maintain original shape by transposing\n    landmarks = [tf.sparse.to_dense(parsed_features[col]) for col in FEATURE_COLUMNS]\n    landmarks = tf.transpose(landmarks)\n    \n    return landmarks, parsed_phrase","metadata":{"execution":{"iopub.status.busy":"2023-08-20T22:24:32.324507Z","iopub.execute_input":"2023-08-20T22:24:32.327087Z","iopub.status.idle":"2023-08-20T22:24:32.336745Z","shell.execute_reply.started":"2023-08-20T22:24:32.327047Z","shell.execute_reply":"2023-08-20T22:24:32.335712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a lookup table using character-to-number mapping\nlookup_table = tf.lookup.StaticHashTable(\n    initializer=tf.lookup.KeyValueTensorInitializer(\n        keys=list(character_mapping.keys()),\n        values=list(character_mapping.values()),\n    ),\n    default_value=tf.constant(-1),\n    name=\"class_mapping_table\"\n)\n\ndef transform_data(landmarks, phrase):\n    # Add start and end tokens to the phrase\n    modified_phrase = start_token + phrase + end_token\n    modified_phrase = tf.strings.bytes_split(modified_phrase)\n    \n    # Lookup the characters in the phrase in the table to convert them to their corresponding numbers\n    modified_phrase = lookup_table.lookup(modified_phrase)\n    \n    # Pad the phrase to ensure all phrases have the same length\n    padded_phrase = tf.pad(modified_phrase, paddings=[[0, 64 - tf.shape(modified_phrase)[0]]], \n                           mode='CONSTANT', constant_values=pad_token_idx)\n    \n    # Pre-process the landmarks\n    processed_landmarks = preprocess_data(landmarks)\n    \n    return processed_landmarks, padded_phrase","metadata":{"execution":{"iopub.status.busy":"2023-08-20T22:24:32.342579Z","iopub.execute_input":"2023-08-20T22:24:32.345498Z","iopub.status.idle":"2023-08-20T22:24:35.521101Z","shell.execute_reply.started":"2023-08-20T22:24:32.345461Z","shell.execute_reply":"2023-08-20T22:24:35.520072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TFRecord files**\n\n\ncreate the training and validation datasets by reading the TFRecord files, mapping the decode and process functions, setting the batch size, and then finally applying prefetching and caching for efficient data loading.\n\n","metadata":{}},{"cell_type":"code","source":"# Define the batch size and compute the size of the training data\nbatch_sz = 64\ntraining_size = int(0.8 * len(tfrecords))\n\n# Construct the training and validation datasets\ntrain_dataset = tf.data.TFRecordDataset(tfrecords[:training_size])\ntrain_dataset = train_dataset.map(parse_record).map(transform_data)\ntrain_dataset = train_dataset.batch(batch_sz).prefetch(buffer_size=tf.data.AUTOTUNE).cache()\n\n\nvalidation_dataset = tf.data.TFRecordDataset(tfrecords[training_size:])\nvalidation_dataset = validation_dataset.map(parse_record).map(transform_data)\nvalidation_dataset = validation_dataset.batch(batch_sz).prefetch(buffer_size=tf.data.AUTOTUNE).cache()\n\n\nprint(\"Number of elements in validation dataset: \", len(list(validation_dataset)))","metadata":{"execution":{"iopub.status.busy":"2023-08-20T22:24:35.522592Z","iopub.execute_input":"2023-08-20T22:24:35.523449Z","iopub.status.idle":"2023-08-20T22:24:53.075979Z","shell.execute_reply.started":"2023-08-20T22:24:35.523414Z","shell.execute_reply":"2023-08-20T22:24:53.074886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Define the Transformer Input Layers**\n\n\n* these custom layers are foundational building blocks for sequence-based models, especially the Transformer architecture.\n* The PhraseEmbedding layer helps to embed a sequence, the LandmarksEmbedding layer captures patterns from landmark data,\n* and the EncoderTransformer processes the embedded input using multi-head attention and feed-forward networks.","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras import layers\n\nclass PhraseEmbedding(layers.Layer):\n    def __init__(self, vocab_size=1000, sequence_length=100, embedding_size=64):\n        super().__init__()\n        self.embedding = tf.keras.layers.Embedding(vocab_size, embedding_size)\n        self.positional_embedding = layers.Embedding(input_dim=sequence_length, output_dim=embedding_size)\n\n    def call(self, inputs):\n        sequence_length = tf.shape(inputs)[-1]\n        positions = tf.range(start=0, limit=sequence_length, delta=1)\n        embedded_inputs = self.embedding(inputs)\n        position_embeddings = self.positional_embedding(positions)\n        return embedded_inputs + position_embeddings\n\n\nclass LandmarksEmbedding(layers.Layer):\n    def __init__(self, embedding_size=64, sequence_length=100):\n        super().__init__()\n        self.convolution1 = tf.keras.layers.Conv1D(\n            embedding_size, 11, strides=2, padding=\"same\", activation=\"relu\"\n        )\n        self.convolution2 = tf.keras.layers.Conv1D(\n            embedding_size, 11, strides=2, padding=\"same\", activation=\"relu\"\n        )\n        self.convolution3 = tf.keras.layers.Conv1D(\n            embedding_size, 11, strides=2, padding=\"same\", activation=\"relu\"\n        )\n        self.positional_embedding = layers.Embedding(input_dim=sequence_length, output_dim=embedding_size)\n\n    def call(self, inputs):\n        inputs = self.convolution1(inputs)\n        inputs = self.convolution2(inputs)\n        return self.convolution3(inputs)","metadata":{"execution":{"iopub.status.busy":"2023-08-20T22:24:53.077731Z","iopub.execute_input":"2023-08-20T22:24:53.078379Z","iopub.status.idle":"2023-08-20T22:24:53.091966Z","shell.execute_reply.started":"2023-08-20T22:24:53.078343Z","shell.execute_reply":"2023-08-20T22:24:53.090772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class EncoderTransformer(layers.Layer):\n    def __init__(self, embedding_dim, heads, ff_dim, dropout_rate=0.1):\n        super().__init__()\n        self.multi_head_attention = layers.MultiHeadAttention(num_heads=heads, key_dim=embedding_dim)\n        self.feed_forward_net = keras.Sequential([\n                layers.Dense(ff_dim, activation=\"relu\"),\n                layers.Dense(embedding_dim),\n            ])\n        self.norm1 = layers.LayerNormalization(epsilon=1e-6)\n        self.norm2 = layers.LayerNormalization(epsilon=1e-6)\n        self.dropout1 = layers.Dropout(dropout_rate)\n        self.dropout2 = layers.Dropout(dropout_rate)\n\n    def call(self, inputs, training):\n        attention_output = self.multi_head_attention(inputs, inputs)\n        attention_output = self.dropout1(attention_output, training=training)\n        temp_out1 = self.norm1(inputs + attention_output)\n        feed_forward_output = self.feed_forward_net(temp_out1)\n        feed_forward_output = self.dropout2(feed_forward_output, training=training)\n        return self.norm2(temp_out1 + feed_forward_output)","metadata":{"execution":{"iopub.status.busy":"2023-08-20T22:24:53.0938Z","iopub.execute_input":"2023-08-20T22:24:53.094181Z","iopub.status.idle":"2023-08-20T22:24:53.104412Z","shell.execute_reply.started":"2023-08-20T22:24:53.094144Z","shell.execute_reply":"2023-08-20T22:24:53.1034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The provided code defines a Transformer-based sequence-to-sequence model.\n* The DecoderTransformer layer processes target sequences in the context of encoder outputs.\n* The TransformerModel represents a full Transformer model and provides methods for training, testing, and inference.\n* The model is designed to handle tasks where a source sequence (like landmarks) is translated into a target sequence (like a phrase).\n\n**DecoderTransformer Layer:**\n\nA custom layer for the decoder part of the Transformer architecture. Contains three layer normalization blocks. Uses multi-head self-attention for processing the target sequence. This self-attention applies a causal mask to prevent future tokens from influencing current token predictions. Uses multi-head encoder-decoder attention, which allows the decoder to focus on relevant parts of the encoder output. Contains a feed-forward neural network applied after the attention mechanisms. Uses dropout for regularization.\n\n**TransformerModel:**\n\nA full Transformer model for sequence-to-sequence tasks. Contains methods for the forward pass, training, testing, and inference. Uses the previously defined LandmarksEmbedding, PhraseEmbedding, EncoderTransformer, and DecoderTransformer layers. The encoder receives source sequences and processes them to produce encoder outputs. The decoder then takes the encoder outputs and target sequences to produce predictions. The training and test steps calculate the model loss and Levenshtein distance. Levenshtein distance measures how many edits are needed to change one sequence into another, and it provides a useful metric for sequence prediction tasks. The inference method predicts a target sequence from a given source sequence.","metadata":{}},{"cell_type":"code","source":"class DecoderTransformer(layers.Layer):\n    def __init__(self, embedding_dimension, num_heads, ff_dimension, dropout_rate=0.1):\n        super(DecoderTransformer, self).__init__()\n        self.layer_norm1 = layers.LayerNormalization(epsilon=1e-6)\n        self.layer_norm2 = layers.LayerNormalization(epsilon=1e-6)\n        self.layer_norm3 = layers.LayerNormalization(epsilon=1e-6)\n        self.self_attention = layers.MultiHeadAttention(num_heads=num_heads, key_dim=embedding_dimension)\n        self.encoder_attention = layers.MultiHeadAttention(num_heads=num_heads, key_dim=embedding_dimension)\n        self.self_dropout = layers.Dropout(0.5)\n        self.encoder_dropout = layers.Dropout(0.1)\n        self.ffn_dropout = layers.Dropout(0.1)\n        self.feed_forward = keras.Sequential([\n            layers.Dense(ff_dimension, activation=\"relu\"),\n            layers.Dense(embedding_dimension),\n        ])\n\n    def create_causal_mask(self, batch_size, n_dest, n_src, dtype):\n        i = tf.range(n_dest)[:, None]\n        j = tf.range(n_src)\n        mask = i >= j - n_src + n_dest\n        mask = tf.cast(mask, dtype)\n        mask = tf.reshape(mask, [1, n_dest, n_src])\n        mult = tf.concat([batch_size[..., tf.newaxis], tf.constant([1, 1], dtype=tf.int32)], 0)\n        return tf.tile(mask, mult)\n\n    def call(self, encoder_output, target, training):\n        target_shape = tf.shape(target)\n        batch_size = target_shape[0]\n        seq_length = target_shape[1]\n        causal_mask = self.create_causal_mask(batch_size, seq_length, seq_length, tf.bool)\n        target_attention = self.self_attention(target, target, attention_mask=causal_mask)\n        target_norm = self.layer_norm1(target + self.self_dropout(target_attention, training=training))\n        encoder_output = self.encoder_attention(target_norm, encoder_output)\n        encoder_norm = self.layer_norm2(self.encoder_dropout(encoder_output, training=training) + target_norm)\n        ffn_output = self.feed_forward(encoder_norm)\n        ffn_norm = self.layer_norm3(encoder_norm + self.ffn_dropout(ffn_output, training=training))\n        return ffn_norm","metadata":{"execution":{"iopub.status.busy":"2023-08-20T22:24:53.109738Z","iopub.execute_input":"2023-08-20T22:24:53.110121Z","iopub.status.idle":"2023-08-20T22:24:53.126047Z","shell.execute_reply.started":"2023-08-20T22:24:53.110094Z","shell.execute_reply":"2023-08-20T22:24:53.124748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow import keras\n\nclass TransformerModel(keras.Model):\n    def __init__(self, hidden_dim=64, num_heads=2, feed_forward_dim=128, \n                 source_max_length=100, target_max_length=100, encoder_layers=4, \n                 decoder_layers=1, num_classes=60):\n        super(TransformerModel, self).__init__()\n        self.loss_tracker = keras.metrics.Mean(name=\"loss\")\n        self.levenshtein_tracker = keras.metrics.Mean(name=\"levenshtein_distance\")\n        \n        self.encoder_layers = encoder_layers\n        self.decoder_layers = decoder_layers\n        self.target_max_length = target_max_length\n        self.num_classes = num_classes\n\n        self.encoder_input = LandmarksEmbedding(hidden_dim, source_max_length)\n        self.decoder_input = PhraseEmbedding(num_classes, target_max_length, hidden_dim)\n\n        self.encoder = keras.Sequential(\n            [self.encoder_input]\n            + [\n                EncoderTransformer(hidden_dim, num_heads, feed_forward_dim)\n                for _ in range(encoder_layers)\n            ]\n        )\n\n        for i in range(decoder_layers):\n            setattr(\n                self,\n                f\"decoder_layer_{i}\",\n                DecoderTransformer(hidden_dim, num_heads, feed_forward_dim),\n            )\n\n        self.classification_layer = layers.Dense(num_classes)\n\n    def decoder_forward_pass(self, encoder_output, target, training):\n        y = self.decoder_input(target)\n        for i in range(self.decoder_layers):\n            y = getattr(self, f\"decoder_layer_{i}\")(encoder_output, y, training)\n        return y\n\n    def call(self, inputs, training):\n        source, target = inputs\n        encoder_output = self.encoder(source, training)\n        decoder_output = self.decoder_forward_pass(encoder_output, target, training)\n        return self.classification_layer(decoder_output)\n\n    @property\n    def metrics(self):\n        return [self.loss_tracker]\n\n    def train_step(self, batch):\n        source, target = batch\n        decoder_input = target[:, :-1]\n        decoder_target = target[:, 1:]\n        \n        with tf.GradientTape() as tape:\n            predictions = self([source, decoder_input])\n            one_hot_targets = tf.one_hot(decoder_target, depth=self.num_classes)\n            mask = tf.math.logical_not(tf.math.equal(decoder_target, pad_token_idx))\n            loss = self.compiled_loss(one_hot_targets, predictions, sample_weight=mask)\n            \n        trainable_vars = self.trainable_variables\n        gradients = tape.gradient(loss, trainable_vars)\n        self.optimizer.apply_gradients(zip(gradients, trainable_vars))\n        \n        levenshtein_distance = tf.edit_distance(tf.sparse.from_dense(target), \n                                                tf.sparse.from_dense(tf.cast(tf.argmax(predictions, axis=1), tf.int32)))\n        levenshtein_distance = tf.reduce_mean(levenshtein_distance)\n        \n        self.levenshtein_tracker.update_state(levenshtein_distance)\n        self.loss_tracker.update_state(loss)\n        return {\"loss\": self.loss_tracker.result(), \"levenshtein_distance\": self.levenshtein_tracker.result()}\n\n    def test_step(self, batch):        \n        source, target = batch\n        decoder_input = target[:, :-1]\n        decoder_target = target[:, 1:]\n        \n        predictions = self([source, decoder_input])\n        one_hot_targets = tf.one_hot(decoder_target, depth=self.num_classes)\n        mask = tf.math.logical_not(tf.math.equal(decoder_target, pad_token_idx))\n        loss = self.compiled_loss(one_hot_targets, predictions, sample_weight=mask)\n        \n        levenshtein_distance = tf.edit_distance(tf.sparse.from_dense(target), \n                                                tf.sparse.from_dense(tf.cast(tf.argmax(predictions, axis=1), tf.int32)))\n        levenshtein_distance = tf.reduce_mean(levenshtein_distance)\n        \n        self.levenshtein_tracker.update_state(levenshtein_distance)\n        self.loss_tracker.update_state(loss)\n        return {\"loss\": self.loss_tracker.result(), \"levenshtein_distance\": self.levenshtein_tracker.result()}\n\n    def inference(self, source, target_start_token_idx):\n        batch_size = tf.shape(source)[0]\n        encoder_output = self.encoder(source, training = False)\n        decoder_input = tf.ones((batch_size, 1), dtype=tf.int32) * target_start_token_idx\n        output_logits = []\n        \n        for _ in range(self.target_max_length - 1):\n            decoder_output = self.decoder_forward_pass(encoder_output, decoder_input, training = False)\n            logits = self.classification_layer(decoder_output)\n            logits = tf.argmax(logits, axis=-1, output_type=tf.int32)\n            last_logit = logits[:, -1][..., tf.newaxis]\n            output_logits.append(last_logit)\n            decoder_input = tf.concat([decoder_input, last_logit], axis=-1)\n        return decoder_input","metadata":{"execution":{"iopub.status.busy":"2023-08-20T22:24:53.128072Z","iopub.execute_input":"2023-08-20T22:24:53.129045Z","iopub.status.idle":"2023-08-20T22:24:53.156491Z","shell.execute_reply.started":"2023-08-20T22:24:53.129004Z","shell.execute_reply":"2023-08-20T22:24:53.155352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**PredictionDisplayCallback:**\n\n* A custom callback derived from Keras' Callbacks.\n\n* Activated at the end of every 4th epoch to display predictions on a batch of test data.\n\n* For each sample in the test batch, it prints out the actual target sequence and the sequence predicted by the model.\n\n* Uses token_index_dict to convert token indices back to characters.\n\n* Sample batch for display:\n\n* Grabs a batch of data from the validation dataset to be used for the PredictionDisplayCallback.\n\n* Model Initialization:\n\n* Instantiates the TransformerModel with specific hyperparameters.\n\n* Learning Rate Scheduler:\n\n* Uses exponential decay for adjusting the learning rate over training steps.\n\n* Loss Function and Optimizer:\n\n* Uses Categorical Crossentropy as the loss function with label smoothing.\n\n* Employs the Adam optimizer with the defined learning rate schedule.\n\n* Checkpointing:\n\n* Sets up a mechanism to save the best model based on validation loss.\n\n* Compile and Train:\n\n* Compiles the model with the specified optimizer and loss function.\n\n* Trains the model on the training dataset, validates on the validation dataset, and uses the custom callback to display predictions after every 4th epoch.","metadata":{}},{"cell_type":"code","source":"class PredictionDisplayCallback(keras.callbacks.Callback):\n    def __init__(self, test_batch, token_index_dict, target_start_token_idx=60, target_end_token_idx=61):\n        \"\"\"\n        Displays predictions for a batch of samples at the end of every 4th epoch.\n\n        Args:\n            test_batch: A batch of test data\n            token_index_dict: A dictionary mapping indices to corresponding tokens in the vocabulary\n            start_token_idx: Index of the start token in the vocabulary\n            end_token_idx: Index of the end token in the vocabulary\n        \"\"\"\n        self.test_batch = test_batch\n        self.target_start_token_idx = target_start_token_idx\n        self.target_end_token_idx = target_end_token_idx\n        self.index_to_token = token_index_dict\n\n    def on_epoch_end(self, epoch, logs=None):\n        if (epoch + 1) % 4 != 0:\n            return\n        source_data = self.test_batch[0]\n        target_data = self.test_batch[1].numpy()\n        batch_size = tf.shape(source_data)[0]\n        predictions = self.model.inference(source_data, self.target_start_token_idx)\n        predictions = predictions.numpy()\n\n        for i in range(batch_size):\n            target_sequence = \"\".join([self.index_to_token[index] for index in target_data[i]])\n            predicted_sequence = \"\"\n            for token_index in predictions[i]:\n                predicted_sequence += self.index_to_token[token_index]\n                if token_index == self.target_end_token_idx:\n                    break\n\n            print(f\"Target sequence:     {target_sequence.replace('-', '')}\")\n            print(f\"Predicted sequence: {predicted_sequence}\\n\")","metadata":{"execution":{"iopub.status.busy":"2023-08-20T22:24:53.15834Z","iopub.execute_input":"2023-08-20T22:24:53.158831Z","iopub.status.idle":"2023-08-20T22:24:53.174709Z","shell.execute_reply.started":"2023-08-20T22:24:53.158795Z","shell.execute_reply":"2023-08-20T22:24:53.173574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sample batch for display\nsample_batch = iter(validation_dataset).next()\n\n# Define mapping from index to characters\nindex_to_char = list(character_mapping.keys())\n\n# Initialize callback\ndisplay_predictions = PredictionDisplayCallback(\n    sample_batch, index_to_char, target_start_token_idx=character_mapping['<'], target_end_token_idx=character_mapping['>']\n)\n\n# Initialize model\nmodel = TransformerModel(\n    hidden_dim=250,\n    num_heads=4,\n    feed_forward_dim=400,\n    source_max_length = FRAME_LEN,\n    target_max_length=64,\n    encoder_layers=2,\n    decoder_layers=1,\n    num_classes=62\n)\n\n# lr_schedule = tf.keras.optimizers.schedules.ExponentialDecay(\n#     0.001, decay_steps=10000, decay_rate=0.9)\n\n# Define loss function\nloss_function = tf.keras.losses.CategoricalCrossentropy(from_logits=True, label_smoothing=0.1)\n\n# Define optimizer\n# optimizer_func = keras.optimizers.Adam(learning_rate=lr_schedule)\noptimizer_func = keras.optimizers.Adam(0.0005)\n\n\n# Compile model\nmodel.compile(optimizer=optimizer_func, loss=loss_function)\n\n# Train model\ntraining_history = model.fit(train_dataset, validation_data=validation_dataset, callbacks=[display_predictions], epochs=15)","metadata":{"execution":{"iopub.status.busy":"2023-08-20T22:53:43.592636Z","iopub.execute_input":"2023-08-20T22:53:43.593798Z","iopub.status.idle":"2023-08-20T23:02:12.753579Z","shell.execute_reply.started":"2023-08-20T22:53:43.593746Z","shell.execute_reply":"2023-08-20T23:02:12.752206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(model.summary())","metadata":{"execution":{"iopub.status.busy":"2023-08-20T23:02:12.756162Z","iopub.execute_input":"2023-08-20T23:02:12.756594Z","iopub.status.idle":"2023-08-20T23:02:12.795737Z","shell.execute_reply.started":"2023-08-20T23:02:12.756538Z","shell.execute_reply":"2023-08-20T23:02:12.794741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Submission**","metadata":{}},{"cell_type":"code","source":"# Import necessary libraries\nimport matplotlib.pyplot as plt\n\n# Create a new figure\nfig, ax = plt.subplots()\n\n# Plot training loss\nax.plot(training_history.history['loss'], label='Train Loss', color='blue')\n\n# Plot validation loss\nax.plot(training_history.history['val_loss'], label='Validation Loss', color='red')\n\n# Set title and labels\nax.set_title('Model Loss During Training')\nax.set_xlabel('Epochs')\nax.set_ylabel('Loss')\n\n# Display legend\nax.legend(loc='upper right')\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-20T23:02:12.796904Z","iopub.execute_input":"2023-08-20T23:02:12.797463Z","iopub.status.idle":"2023-08-20T23:02:13.099571Z","shell.execute_reply.started":"2023-08-20T23:02:12.797426Z","shell.execute_reply":"2023-08-20T23:02:13.098619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create TFLite model\n    \nclass TFLiteModel(tf.Module):\n    def __init__(self, model):\n        super(TFLiteModel, self).__init__()\n        self.target_start_token_idx = start_token_idx\n        self.target_end_token_idx = end_token_idx\n        # Load the feature generation and main models\n        self.model = model\n    \n    @tf.function(input_signature=[tf.TensorSpec(shape=[None, len(FEATURE_COLUMNS)], dtype=tf.float32, name='inputs')])\n    def __call__(self, inputs, training=False):\n        # Preprocess Data\n        x = tf.cast(inputs, tf.float32)\n        x = x[None]\n        x = tf.cond(tf.shape(x)[1] == 0, lambda: tf.zeros((1, 1, len(FEATURE_COLUMNS))), lambda: tf.identity(x))\n        x = x[0]\n        x = preprocess_data(x)\n        x = x[None]\n        x = self.model.inference(x, self.target_start_token_idx)\n        x = x[0]\n        idx = tf.argmax(tf.cast(tf.equal(x, self.target_end_token_idx), tf.int32))\n        idx = tf.where(tf.math.less(idx, 1), tf.constant(2, dtype=tf.int64), idx)\n        x = x[1:idx]\n        x = tf.one_hot(x, 59)\n        return {'outputs': x}\n    \ntflitemodel_base = TFLiteModel(model)","metadata":{"execution":{"iopub.status.busy":"2023-08-20T23:02:13.10241Z","iopub.execute_input":"2023-08-20T23:02:13.103008Z","iopub.status.idle":"2023-08-20T23:02:13.114902Z","shell.execute_reply.started":"2023-08-20T23:02:13.10297Z","shell.execute_reply":"2023-08-20T23:02:13.11359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save_weights(\"model.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-08-20T23:02:13.116444Z","iopub.execute_input":"2023-08-20T23:02:13.117654Z","iopub.status.idle":"2023-08-20T23:02:13.268638Z","shell.execute_reply.started":"2023-08-20T23:02:13.117615Z","shell.execute_reply":"2023-08-20T23:02:13.26728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras_model_converter = tf.lite.TFLiteConverter.from_keras_model(tflitemodel_base)\nkeras_model_converter.target_spec.supported_ops = [tf.lite.OpsSet.TFLITE_BUILTINS]\ntflite_model = keras_model_converter.convert()\nwith open('/kaggle/working/model.tflite', 'wb') as f:\n    f.write(tflite_model)\n    \ninfargs = {\"selected_columns\" : FEATURE_COLUMNS}\n\nwith open('inference_args.json', \"w\") as json_file:\n    json.dump(infargs, json_file)","metadata":{"execution":{"iopub.status.busy":"2023-08-20T23:03:11.525605Z","iopub.execute_input":"2023-08-20T23:03:11.526094Z","iopub.status.idle":"2023-08-20T23:03:58.579659Z","shell.execute_reply.started":"2023-08-20T23:03:11.526054Z","shell.execute_reply":"2023-08-20T23:03:58.578516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip submission.zip  './model.tflite' './inference_args.json'","metadata":{"execution":{"iopub.status.busy":"2023-08-20T23:03:58.581571Z","iopub.execute_input":"2023-08-20T23:03:58.581987Z","iopub.status.idle":"2023-08-20T23:04:01.161754Z","shell.execute_reply.started":"2023-08-20T23:03:58.58195Z","shell.execute_reply":"2023-08-20T23:04:01.160433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interpreter = tf.lite.Interpreter(\"model.tflite\")\n\nREQUIRED_SIGNATURE = \"serving_default\"\nREQUIRED_OUTPUT = \"outputs\"\n\nwith open (\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\", \"r\") as f:\n    character_map = json.load(f)\nrev_character_map = {j:i for i,j in character_map.items()}\n\nfound_signatures = list(interpreter.get_signature_list().keys())\n\nif REQUIRED_SIGNATURE not in found_signatures:\n    raise KernelEvalException('Required input signature not found.')\n\nprediction_fn = interpreter.get_signature_runner(\"serving_default\")\noutput = prediction_fn(inputs=sample_batch[0][0])\nprediction_str = \"\".join([rev_character_map.get(s, \"\") for s in np.argmax(output[REQUIRED_OUTPUT], axis=1)])\nprint(prediction_str)","metadata":{"execution":{"iopub.status.busy":"2023-08-20T23:04:01.163966Z","iopub.execute_input":"2023-08-20T23:04:01.164643Z","iopub.status.idle":"2023-08-20T23:04:01.558026Z","shell.execute_reply.started":"2023-08-20T23:04:01.1646Z","shell.execute_reply":"2023-08-20T23:04:01.556898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"References:\n\nhttps://www.kaggle.com/code/gusthema/asl-fingerspelling-recognition-w-tensorflow#Train-the-Transformer-model\n\nhttps://keras.io/examples/\n\nhttps://www.kaggle.com/code/shlomoron/aslfr-a-simple-transformer/notebook#TFLiteModel","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}