{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n!pip install mediapipe\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\"\"\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-07-28T09:48:09.369389Z","iopub.execute_input":"2023-07-28T09:48:09.369681Z","iopub.status.idle":"2023-07-28T09:48:25.518239Z","shell.execute_reply.started":"2023-07-28T09:48:09.369654Z","shell.execute_reply":"2023-07-28T09:48:25.517023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\nimport numpy as np\nimport pandas as pd\n# import pyarrow.parquet as pq\nimport tensorflow as tf\nimport json\n# import mediapipe\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport random\n\n# from skimage.transform import resize\n# from mediapipe.framework.formats import landmark_pb2\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tqdm.notebook import tqdm\nfrom tqdm import tqdm as TQ\nfrom matplotlib import animation, rc","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:25.520206Z","iopub.execute_input":"2023-07-28T09:48:25.520527Z","iopub.status.idle":"2023-07-28T09:48:33.997131Z","shell.execute_reply.started":"2023-07-28T09:48:25.5205Z","shell.execute_reply":"2023-07-28T09:48:33.996069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:33.998799Z","iopub.execute_input":"2023-07-28T09:48:33.999879Z","iopub.status.idle":"2023-07-28T09:48:34.178737Z","shell.execute_reply.started":"2023-07-28T09:48:33.999835Z","shell.execute_reply":"2023-07-28T09:48:34.17768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LPOSE = [13, 15, 17, 19, 21]\nRPOSE = [14, 16, 18, 20, 22]\nPOSE = LPOSE + RPOSE","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:34.181928Z","iopub.execute_input":"2023-07-28T09:48:34.182652Z","iopub.status.idle":"2023-07-28T09:48:34.188732Z","shell.execute_reply.started":"2023-07-28T09:48:34.182616Z","shell.execute_reply":"2023-07-28T09:48:34.187721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = [f'x_right_hand_{i}' for i in range(21)] + [f'x_left_hand_{i}' for i in range(21)] + [f'x_pose_{i}' for i in POSE]\nY = [f'y_right_hand_{i}' for i in range(21)] + [f'y_left_hand_{i}' for i in range(21)] + [f'y_pose_{i}' for i in POSE]\nZ = [f'z_right_hand_{i}' for i in range(21)] + [f'z_left_hand_{i}' for i in range(21)] + [f'z_pose_{i}' for i in POSE]\nFEATURE_COLUMNS = X + Y + Z","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:34.19341Z","iopub.execute_input":"2023-07-28T09:48:34.194339Z","iopub.status.idle":"2023-07-28T09:48:34.20663Z","shell.execute_reply.started":"2023-07-28T09:48:34.194303Z","shell.execute_reply":"2023-07-28T09:48:34.20562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"x_\" in col]\nY_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"y_\" in col]\nZ_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"z_\" in col]\n\nRHAND_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"right\" in col]\nLHAND_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"left\" in col]\nRPOSE_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"pose\" in col and int(col[-2:]) in RPOSE]\nLPOSE_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"pose\" in col and int(col[-2:]) in LPOSE]","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:34.211224Z","iopub.execute_input":"2023-07-28T09:48:34.213305Z","iopub.status.idle":"2023-07-28T09:48:34.226262Z","shell.execute_reply.started":"2023-07-28T09:48:34.21327Z","shell.execute_reply":"2023-07-28T09:48:34.225206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open (\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\", \"r\") as f:\n    char_to_num = json.load(f)\n\n# Add pad_token, start pointer and end pointer to the dict\npad_token = 'P'\nstart_token = '<'\nend_token = '>'\nLPOSE = [13, 15, 17, 19, 21]\nRPOSE = [14, 16, 18, 20, 22]\nPOSE = LPOSE + RPOSE\npad_token_idx = 59\nstart_token_idx = 60\nend_token_idx = 61\n\nchar_to_num[pad_token] = pad_token_idx\nchar_to_num[start_token] = start_token_idx\nchar_to_num[end_token] = end_token_idx\nnum_to_char = {j:i for i,j in char_to_num.items()}","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:34.231908Z","iopub.execute_input":"2023-07-28T09:48:34.234799Z","iopub.status.idle":"2023-07-28T09:48:34.249504Z","shell.execute_reply.started":"2023-07-28T09:48:34.23476Z","shell.execute_reply":"2023-07-28T09:48:34.248204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set length of frames to 128\nFRAME_LEN = 128\n\n# Function to resize and add padding.\ndef resize_pad(x):\n    if tf.shape(x)[0] < FRAME_LEN:\n        x = tf.pad(x, ([[0, FRAME_LEN-tf.shape(x)[0]], [0, 0], [0, 0]]))\n    else:\n        x = tf.image.resize(x, (FRAME_LEN, tf.shape(x)[1]))\n    return x\n\n# Detect the dominant hand from the number of NaN values.\n# Dominant hand will have less NaN values since it is in frame moving.\ndef pre_process(x):\n    rhand = tf.gather(x, RHAND_IDX, axis=1)\n    lhand = tf.gather(x, LHAND_IDX, axis=1)\n    rpose = tf.gather(x, RPOSE_IDX, axis=1)\n    lpose = tf.gather(x, LPOSE_IDX, axis=1)\n    \n    rnan_idx = tf.reduce_any(tf.math.is_nan(rhand), axis=1)\n    lnan_idx = tf.reduce_any(tf.math.is_nan(lhand), axis=1)\n    \n    rnans = tf.math.count_nonzero(rnan_idx)\n    lnans = tf.math.count_nonzero(lnan_idx)\n    \n    # For dominant hand\n    if rnans > lnans:\n        hand = lhand\n        pose = lpose\n        \n        hand_x = hand[:, 0*(len(LHAND_IDX)//3) : 1*(len(LHAND_IDX)//3)]\n        hand_y = hand[:, 1*(len(LHAND_IDX)//3) : 2*(len(LHAND_IDX)//3)]\n        hand_z = hand[:, 2*(len(LHAND_IDX)//3) : 3*(len(LHAND_IDX)//3)]\n        hand = tf.concat([1-hand_x, hand_y, hand_z], axis=1)\n        \n        pose_x = pose[:, 0*(len(LPOSE_IDX)//3) : 1*(len(LPOSE_IDX)//3)]\n        pose_y = pose[:, 1*(len(LPOSE_IDX)//3) : 2*(len(LPOSE_IDX)//3)]\n        pose_z = pose[:, 2*(len(LPOSE_IDX)//3) : 3*(len(LPOSE_IDX)//3)]\n        pose = tf.concat([1-pose_x, pose_y, pose_z], axis=1)\n    else:\n        hand = rhand\n        pose = rpose\n    \n        hand_x = hand[:, 0*(len(RHAND_IDX)//3) : 1*(len(RHAND_IDX)//3)]\n        hand_y = hand[:, 1*(len(RHAND_IDX)//3) : 2*(len(RHAND_IDX)//3)]\n        hand_z = hand[:, 2*(len(RHAND_IDX)//3) : 3*(len(RHAND_IDX)//3)]\n        hand = tf.concat([hand_x, hand_y, hand_z], axis=1)\n        \n        pose_x = pose[:, 0*(len(RPOSE_IDX)//3) : 1*(len(RPOSE_IDX)//3)]\n        pose_y = pose[:, 1*(len(RPOSE_IDX)//3) : 2*(len(RPOSE_IDX)//3)]\n        pose_z = pose[:, 2*(len(RPOSE_IDX)//3) : 3*(len(RPOSE_IDX)//3)]\n        pose = tf.concat([pose_x, pose_y, pose_z], axis=1)\n        \n        \n    hand = tf.concat([hand_x[..., tf.newaxis], hand_y[..., tf.newaxis], hand_z[..., tf.newaxis]], axis=-1)\n    \n    mean = tf.math.reduce_mean(hand, axis=1)[:, tf.newaxis, :]\n    std = tf.math.reduce_std(hand, axis=1)[:, tf.newaxis, :]\n    hand = (hand - mean) / std\n\n    pose = tf.concat([pose_x[..., tf.newaxis], pose_y[..., tf.newaxis], pose_z[..., tf.newaxis]], axis=-1)\n    \n    x = tf.concat([hand, pose], axis=1)\n    x = resize_pad(x)\n    \n    x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n    x = tf.reshape(x, (FRAME_LEN, len(LHAND_IDX) + len(LPOSE_IDX)))\n    return x\n","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:34.255062Z","iopub.execute_input":"2023-07-28T09:48:34.25785Z","iopub.status.idle":"2023-07-28T09:48:34.288842Z","shell.execute_reply.started":"2023-07-28T09:48:34.257813Z","shell.execute_reply":"2023-07-28T09:48:34.287655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# To Parse data from tfrecord","metadata":{}},{"cell_type":"code","source":"def decode_fn(record_bytes):\n    schema = {COL: tf.io.VarLenFeature(dtype=tf.float32) for COL in FEATURE_COLUMNS}\n    schema[\"phrase\"] = tf.io.FixedLenFeature([], dtype=tf.string)\n    features = tf.io.parse_single_example(record_bytes, schema)\n    phrase = features[\"phrase\"]\n    landmarks = ([tf.sparse.to_dense(features[COL]) for COL in FEATURE_COLUMNS])\n    # Transpose to maintain the original shape of landmarks data.\n    landmarks = tf.transpose(landmarks)\n    \n    return landmarks, phrase","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:34.290849Z","iopub.execute_input":"2023-07-28T09:48:34.291838Z","iopub.status.idle":"2023-07-28T09:48:34.307809Z","shell.execute_reply.started":"2023-07-28T09:48:34.291802Z","shell.execute_reply":"2023-07-28T09:48:34.306804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# To Convert Data","metadata":{}},{"cell_type":"code","source":"table = tf.lookup.StaticHashTable(\n    initializer=tf.lookup.KeyValueTensorInitializer(\n        keys=list(char_to_num.keys()),\n        values=list(char_to_num.values()),\n    ),\n    default_value=tf.constant(-1),\n    name=\"class_weight\"\n)\n\ndef convert_fn(landmarks, phrase):\n    # Add start and end pointers to phrase.\n    phrase = start_token + phrase + end_token\n    phrase = tf.strings.bytes_split(phrase)\n    phrase = table.lookup(phrase)\n    # Vectorize and add padding.\n    phrase = tf.pad(phrase, paddings=[[0, 64 - tf.shape(phrase)[0]]], mode = 'CONSTANT',\n                    constant_values = pad_token_idx)\n    # Apply pre_process function to the landmarks.\n    return pre_process(landmarks), phrase","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:34.31315Z","iopub.execute_input":"2023-07-28T09:48:34.313951Z","iopub.status.idle":"2023-07-28T09:48:37.265253Z","shell.execute_reply.started":"2023-07-28T09:48:34.313916Z","shell.execute_reply":"2023-07-28T09:48:37.26413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Map the tfrecord files into a variable","metadata":{}},{"cell_type":"code","source":"tf_records = dataset_df.file_id.map(lambda x: f'/kaggle/input/aslfr-parquets-to-tfrecords-cleaned/tfds/{x}.tfrecord').unique()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:37.266782Z","iopub.execute_input":"2023-07-28T09:48:37.267153Z","iopub.status.idle":"2023-07-28T09:48:37.331994Z","shell.execute_reply.started":"2023-07-28T09:48:37.267118Z","shell.execute_reply":"2023-07-28T09:48:37.330941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\n# Assuming you have the necessary functions 'decode_fn' and 'convert_fn' defined for parsing and preprocessing the TFRecords.\n\nbatch_size = 64\ntrain_len = int(0.8 * len(tf_records))\nval_len = int(0.1 * len(tf_records))\n\n# Training dataset\ntrain_ds = tf.data.TFRecordDataset(tf_records[:train_len])\ntrain_ds = train_ds.map(decode_fn)\ntrain_ds = train_ds.map(convert_fn)\ntrain_ds = train_ds.batch(batch_size)\ntrain_ds = train_ds.prefetch(tf.data.AUTOTUNE)\ntrain_ds = train_ds.cache()\n\n# Validation dataset\nvalid_ds = tf.data.TFRecordDataset(tf_records[train_len:train_len+val_len])\nvalid_ds = valid_ds.map(decode_fn)\nvalid_ds = valid_ds.map(convert_fn)\nvalid_ds = valid_ds.batch(batch_size)\nvalid_ds = valid_ds.prefetch(tf.data.AUTOTUNE)\nvalid_ds = valid_ds.cache()\n\n# Test dataset\ntest_ds = tf.data.TFRecordDataset(tf_records[train_len+val_len:])\ntest_ds = test_ds.map(decode_fn)\ntest_ds = test_ds.map(convert_fn)\ntest_ds = test_ds.batch(batch_size)\ntest_ds = test_ds.prefetch(tf.data.AUTOTUNE)\ntest_ds = test_ds.cache()\n\n# batch_size = 64\n# train_len = int(0.8 * len(tf_records))\n\n\n# train_ds = tf.data.TFRecordDataset(tf_records[:train_len]).map(decode_fn).map(convert_fn).batch(batch_size).prefetch(buffer_size=tf.data.AUTOTUNE).cache()\n# valid_ds = tf.data.TFRecordDataset(tf_records[train_len:]).map(decode_fn).map(convert_fn).batch(batch_size).prefetch(buffer_size=tf.data.AUTOTUNE).cache()\n\n\n\n# batch_size = 64\n\n# # Assuming X_train, y_train, X_valid, and y_valid are your NumPy arrays\n# # Convert the NumPy arrays to TensorFlow datasets\n# train_ds = tf.data.Dataset.from_tensor_slices((X_train, y_train))\n# valid_ds = tf.data.Dataset.from_tensor_slices((X_valid, y_valid))\n\n# # Optionally, you can shuffle the training dataset and repeat it for multiple epochs\n# train_ds = train_ds.shuffle(len(X_train)).repeat()\n\n# # Batch and prefetch the datasets\n# train_ds = train_ds.batch(batch_size).prefetch(buffer_size=tf.data.AUTOTUNE)\n# valid_ds = valid_ds.batch(batch_size).prefetch(buffer_size=tf.data.AUTOTUNE)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:37.333432Z","iopub.execute_input":"2023-07-28T09:48:37.334038Z","iopub.status.idle":"2023-07-28T09:48:39.82986Z","shell.execute_reply.started":"2023-07-28T09:48:37.333998Z","shell.execute_reply":"2023-07-28T09:48:39.828743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nvalid_ds","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:39.83141Z","iopub.execute_input":"2023-07-28T09:48:39.832301Z","iopub.status.idle":"2023-07-28T09:48:39.847685Z","shell.execute_reply.started":"2023-07-28T09:48:39.832235Z","shell.execute_reply":"2023-07-28T09:48:39.846347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TokenEmbedding(layers.Layer):\n    def __init__(self, num_vocab=1000, maxlen=100, num_hid=64):\n        super().__init__()\n        self.emb = tf.keras.layers.Embedding(num_vocab, num_hid)\n        self.pos_emb = layers.Embedding(input_dim=maxlen, output_dim=num_hid)\n\n    def call(self, x):\n        maxlen = tf.shape(x)[-1]\n        x = self.emb(x)\n        positions = tf.range(start=0, limit=maxlen, delta=1)\n        positions = self.pos_emb(positions)\n        return x + positions\n\n\nclass LandmarkEmbedding(layers.Layer):\n    def __init__(self, num_hid=64, maxlen=100):\n        super().__init__()\n        self.conv1 = tf.keras.layers.Conv1D(\n            num_hid, 11, strides=2, padding=\"same\", activation=\"relu\"\n        )\n        self.conv2 = tf.keras.layers.Conv1D(\n            num_hid, 11, strides=2, padding=\"same\", activation=\"relu\"\n        )\n        self.conv3 = tf.keras.layers.Conv1D(\n            num_hid, 11, strides=2, padding=\"same\", activation=\"relu\"\n        )\n        self.pos_emb = layers.Embedding(input_dim=maxlen, output_dim=num_hid)\n\n    def call(self, x):\n        x = self.conv1(x)\n        x = self.conv2(x)\n        return self.conv3(x)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:39.848919Z","iopub.execute_input":"2023-07-28T09:48:39.849268Z","iopub.status.idle":"2023-07-28T09:48:39.866868Z","shell.execute_reply.started":"2023-07-28T09:48:39.849241Z","shell.execute_reply":"2023-07-28T09:48:39.865793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class LSTM_Encoder(layers.Layer):\n    def __init__(self, embed_dim, num_heads, feed_forward_dim, rate=0.1):\n        super().__init__()  # Correct usage of super() with parentheses\n        self.enc_lstm_layer = keras.Sequential([\n            layers.LSTM(4,\n                        activation='tanh',\n                        return_sequences=True,\n                        recurrent_initializer='glorot_uniform')\n        ])\n        self.att = layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        self.enc_layernorm = layers.LayerNormalization(epsilon=1e-6)\n\n        self.ffn = keras.Sequential([\n            layers.Dense(feed_forward_dim, activation='relu'),\n            layers.Dense(embed_dim),\n        ])\n\n        self.dropout1 = layers.Dropout(rate)\n        self.dropout2 = layers.Dropout(rate)\n\n\n    \n    def call(self, inputs, training=True):  # Pass 'inputs' and set 'training' default to True\n        lstm_outputs = self.enc_lstm_layer(inputs)  # Pass 'inputs' to the LSTM layer\n        lstm_outputs = self.dropout1(lstm_outputs, training=training)  # Apply dropout during training\n        ffn_output = self.ffn(lstm_outputs)\n        ffn_output = self.dropout2(ffn_output, training=training)  # Apply dropout during training\n        return self.enc_layernorm(ffn_output + inputs)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:39.86857Z","iopub.execute_input":"2023-07-28T09:48:39.868906Z","iopub.status.idle":"2023-07-28T09:48:39.879605Z","shell.execute_reply.started":"2023-07-28T09:48:39.868873Z","shell.execute_reply":"2023-07-28T09:48:39.878745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TransformerEncoder(layers.Layer):\n    def __init__(self, embed_dim, num_heads, feed_forward_dim, rate=0.1):\n        super().__init__()\n        self.att = layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        self.ffn = keras.Sequential(\n            [\n                \n                layers.Dense(feed_forward_dim, activation=\"relu\"),\n                layers.Dense(embed_dim),\n            ]\n        )\n        self.layernorm1 = layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm2 = layers.LayerNormalization(epsilon=1e-6)\n        self.dropout1 = layers.Dropout(rate)\n        self.dropout2 = layers.Dropout(rate)\n\n    def call(self, inputs, training):\n        attn_output = self.att(inputs, inputs)\n        attn_output = self.dropout1(attn_output, training=training)\n        out1 = self.layernorm1(inputs + attn_output)\n        ffn_output = self.ffn(out1)\n        ffn_output = self.dropout2(ffn_output, training=training)\n        return self.layernorm2(out1 + ffn_output)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:39.881345Z","iopub.execute_input":"2023-07-28T09:48:39.88178Z","iopub.status.idle":"2023-07-28T09:48:39.89524Z","shell.execute_reply.started":"2023-07-28T09:48:39.881745Z","shell.execute_reply":"2023-07-28T09:48:39.89429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TransformerDecoder(layers.Layer):\n    def __init__(self, embed_dim, num_heads, feed_forward_dim, dropout_rate=0.1):\n        super().__init__()\n        self.layernorm1 = layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm2 = layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm3 = layers.LayerNormalization(epsilon=1e-6)\n        self.self_att = layers.MultiHeadAttention(\n            num_heads=num_heads, key_dim=embed_dim\n        )\n        self.enc_att = layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        self.self_dropout = layers.Dropout(0.5)\n        self.enc_dropout = layers.Dropout(0.1)\n        self.ffn_dropout = layers.Dropout(0.1)\n        self.ffn = keras.Sequential(\n            [\n                layers.Dense(feed_forward_dim, activation=\"relu\"),\n                layers.Dense(embed_dim),\n            ]\n        )\n\n    def causal_attention_mask(self, batch_size, n_dest, n_src, dtype):\n        \"\"\"Masks the upper half of the dot product matrix in self attention.\n\n        This prevents flow of information from future tokens to current token.\n        1's in the lower triangle, counting from the lower right corner.\n        \"\"\"\n        i = tf.range(n_dest)[:, None]\n        j = tf.range(n_src)\n        m = i >= j - n_src + n_dest\n        mask = tf.cast(m, dtype)\n        mask = tf.reshape(mask, [1, n_dest, n_src])\n        mult = tf.concat(\n            [batch_size[..., tf.newaxis], tf.constant([1, 1], dtype=tf.int32)], 0\n        )\n        return tf.tile(mask, mult)\n\n    def call(self, enc_out, target, training):\n        input_shape = tf.shape(target)\n        batch_size = input_shape[0]\n        seq_len = input_shape[1]\n        causal_mask = self.causal_attention_mask(batch_size, seq_len, seq_len, tf.bool)\n        target_att = self.self_att(target, target, attention_mask=causal_mask)\n        target_norm = self.layernorm1(target + self.self_dropout(target_att, training = training))\n        enc_out = self.enc_att(target_norm, enc_out)\n        enc_out_norm = self.layernorm2(self.enc_dropout(enc_out, training = training) + target_norm)\n        ffn_out = self.ffn(enc_out_norm)\n        ffn_out_norm = self.layernorm3(enc_out_norm + self.ffn_dropout(ffn_out, training = training))\n        return ffn_out_norm","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:39.897053Z","iopub.execute_input":"2023-07-28T09:48:39.897504Z","iopub.status.idle":"2023-07-28T09:48:39.913887Z","shell.execute_reply.started":"2023-07-28T09:48:39.897462Z","shell.execute_reply":"2023-07-28T09:48:39.913025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Customized to add edit_dist metric and training variable.\n# Reference:\n# https://www.kaggle.com/code/irohith/aslfr-transformer/notebook\n# https://www.kaggle.com/code/shlomoron/aslfr-a-simple-transformer/notebook\n\nclass Transformer(keras.Model):\n    def __init__(\n        self,\n        num_hid=64,\n        num_head=2,\n        num_feed_forward=128,\n        source_maxlen=100,\n        target_maxlen=100,\n        num_layers_enc=4,\n        num_layers_dec=1,\n        num_classes=60,\n    ):\n        super().__init__()\n        self.loss_metric = keras.metrics.Mean(name=\"loss\")\n        self.acc_metric = keras.metrics.Mean(name=\"edit_dist\")\n        self.num_layers_enc = num_layers_enc\n        self.num_layers_dec = num_layers_dec\n        self.target_maxlen = target_maxlen\n        self.num_classes = num_classes\n\n        self.enc_input = LandmarkEmbedding(num_hid=num_hid, maxlen=source_maxlen)\n        self.dec_input = TokenEmbedding(\n            num_vocab=num_classes, maxlen=target_maxlen, num_hid=num_hid\n        )\n\n        self.encoder = keras.Sequential(\n            [self.enc_input]\n            + [\n                LSTM_Encoder(num_hid, num_head, num_feed_forward)\n                for _ in range(num_layers_enc)\n            ]\n        )\n\n        for i in range(num_layers_dec):\n            setattr(\n                self,\n                f\"dec_layer_{i}\",\n                TransformerDecoder(num_hid, num_head, num_feed_forward),\n            )\n\n        self.classifier = layers.Dense(num_classes)\n\n    def decode(self, enc_out, target, training):\n        y = self.dec_input(target)\n        for i in range(self.num_layers_dec):\n            y = getattr(self, f\"dec_layer_{i}\")(enc_out, y, training)\n        return y\n\n    def call(self, inputs, training):\n        source = inputs[0]\n        target = inputs[1]\n        x = self.encoder(source, training)\n        y = self.decode(x, target, training)\n        return self.classifier(y)\n\n    @property\n    def metrics(self):\n        return [self.loss_metric]\n\n    def train_step(self, batch):\n        \"\"\"Processes one batch inside model.fit().\"\"\"\n        source = batch[0]\n        target = batch[1]\n\n        input_shape = tf.shape(target)\n        batch_size = input_shape[0]\n        \n        dec_input = target[:, :-1]\n        dec_target = target[:, 1:]\n        with tf.GradientTape() as tape:\n            preds = self([source, dec_input])\n            one_hot = tf.one_hot(dec_target, depth=self.num_classes)\n            mask = tf.math.logical_not(tf.math.equal(dec_target, pad_token_idx))\n            loss = self.compiled_loss(one_hot, preds, sample_weight=mask)\n        trainable_vars = self.trainable_variables\n        gradients = tape.gradient(loss, trainable_vars)\n        self.optimizer.apply_gradients(zip(gradients, trainable_vars))\n        # Computes the Levenshtein distance between sequences since the evaluation\n        # metric for this contest is the normalized total levenshtein distance.\n        edit_dist = tf.edit_distance(tf.sparse.from_dense(target), \n                                     tf.sparse.from_dense(tf.cast(tf.argmax(preds, axis=1), tf.int32)))\n        edit_dist = tf.reduce_mean(edit_dist)\n        self.acc_metric.update_state(edit_dist)\n        self.loss_metric.update_state(loss)\n        return {\"loss\": self.loss_metric.result(), \"edit_dist\": self.acc_metric.result()}\n\n    def test_step(self, batch):        \n        source = batch[0]\n        target = batch[1]\n\n        input_shape = tf.shape(target)\n        batch_size = input_shape[0]\n        \n        dec_input = target[:, :-1]\n        dec_target = target[:, 1:]\n        preds = self([source, dec_input])\n        one_hot = tf.one_hot(dec_target, depth=self.num_classes)\n        mask = tf.math.logical_not(tf.math.equal(dec_target, pad_token_idx))\n        loss = self.compiled_loss(one_hot, preds, sample_weight=mask)\n        # Computes the Levenshtein distance between sequences since the evaluation\n        # metric for this contest is the normalized total levenshtein distance.\n        edit_dist = tf.edit_distance(tf.sparse.from_dense(target), \n                                     tf.sparse.from_dense(tf.cast(tf.argmax(preds, axis=1), tf.int32)))\n        edit_dist = tf.reduce_mean(edit_dist)\n        self.acc_metric.update_state(edit_dist)\n        self.loss_metric.update_state(loss)\n        return {\"loss\": self.loss_metric.result(), \"edit_dist\": self.acc_metric.result()}\n\n    def generate(self, source, target_start_token_idx):\n        \"\"\"Performs inference over one batch of inputs using greedy decoding.\"\"\"\n        bs = tf.shape(source)[0]\n        enc = self.encoder(source, training = False)\n        dec_input = tf.ones((bs, 1), dtype=tf.int32) * target_start_token_idx\n        dec_logits = []\n        for i in range(self.target_maxlen - 1):\n            dec_out = self.decode(enc, dec_input, training = False)\n            logits = self.classifier(dec_out)\n            logits = tf.argmax(logits, axis=-1, output_type=tf.int32)\n            last_logit = logits[:, -1][..., tf.newaxis]\n            dec_logits.append(last_logit)\n            dec_input = tf.concat([dec_input, last_logit], axis=-1)\n        return dec_input","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:39.916893Z","iopub.execute_input":"2023-07-28T09:48:39.91718Z","iopub.status.idle":"2023-07-28T09:48:39.942376Z","shell.execute_reply.started":"2023-07-28T09:48:39.917156Z","shell.execute_reply":"2023-07-28T09:48:39.941421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DisplayOutputs(keras.callbacks.Callback):\n    def __init__(\n        self, batch, idx_to_token, target_start_token_idx=60, target_end_token_idx=61\n    ):\n        \"\"\"Displays a batch of outputs after every 4 epoch\n\n        Args:\n            batch: A test batch\n            idx_to_token: A List containing the vocabulary tokens corresponding to their indices\n            target_start_token_idx: A start token index in the target vocabulary\n            target_end_token_idx: An end token index in the target vocabulary\n        \"\"\"\n        self.batch = batch\n        self.target_start_token_idx = target_start_token_idx\n        self.target_end_token_idx = target_end_token_idx\n        self.idx_to_char = idx_to_token\n\n    def on_epoch_end(self, epoch, logs=None):\n        if epoch % 4 != 0:\n            return\n        source = self.batch[0]\n        target = self.batch[1].numpy()\n        bs = tf.shape(source)[0]\n        preds = self.model.generate(source, self.target_start_token_idx)\n        preds = preds.numpy()\n        for i in range(bs):\n            target_text = \"\".join([self.idx_to_char[_] for _ in target[i, :]])\n            prediction = \"\"\n            for idx in preds[i, :]:\n                prediction += self.idx_to_char[idx]\n                if idx == self.target_end_token_idx:\n                    break\n            print(f\"target:     {target_text.replace('-','')}\")\n            print(f\"prediction: {prediction}\\n\")","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:39.943859Z","iopub.execute_input":"2023-07-28T09:48:39.944182Z","iopub.status.idle":"2023-07-28T09:48:39.958238Z","shell.execute_reply.started":"2023-07-28T09:48:39.944151Z","shell.execute_reply":"2023-07-28T09:48:39.957282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transformer variables are customized from original keras tutorial to suit this dataset.\n# Reference: https://www.kaggle.com/code/shlomoron/aslfr-a-simple-transformer/notebook\n\nbatch = next(iter(valid_ds))\n\n# The vocabulary to convert predicted indices into characters\nidx_to_char = list(char_to_num.keys())\ndisplay_cb = DisplayOutputs(\n    batch, idx_to_char, target_start_token_idx=char_to_num['<'], target_end_token_idx=char_to_num['>']\n)  # set the arguments as per vocabulary index for '<' and '>'\n\nmodel = Transformer(\n    num_hid=200,\n    num_head=4,\n    num_feed_forward=400,\n    source_maxlen = FRAME_LEN,\n    target_maxlen=64,\n    num_layers_enc=2,\n    num_layers_dec=1,\n    num_classes=62\n)\nloss_fn = tf.keras.losses.CategoricalCrossentropy(\n    from_logits=True, label_smoothing=0.1,\n)\n\n\noptimizer = keras.optimizers.Adam(0.0001)\nmodel.compile(optimizer=optimizer, loss=loss_fn,metrics=['accuracy'])\n\n\nhistory = model.fit(train_ds, validation_data=valid_ds, callbacks=[display_cb], epochs=13)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:48:39.961296Z","iopub.execute_input":"2023-07-28T09:48:39.961667Z","iopub.status.idle":"2023-07-28T09:56:37.820217Z","shell.execute_reply.started":"2023-07-28T09:48:39.961617Z","shell.execute_reply":"2023-07-28T09:56:37.819066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-07-28T07:07:19.555725Z","iopub.execute_input":"2023-07-28T07:07:19.556114Z","iopub.status.idle":"2023-07-28T07:07:19.566494Z","shell.execute_reply.started":"2023-07-28T07:07:19.556083Z","shell.execute_reply":"2023-07-28T07:07:19.565173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.legend(['training loss', 'val_loss'])","metadata":{"execution":{"iopub.status.busy":"2023-07-28T07:40:59.454728Z","iopub.execute_input":"2023-07-28T07:40:59.455142Z","iopub.status.idle":"2023-07-28T07:40:59.785522Z","shell.execute_reply.started":"2023-07-28T07:40:59.455111Z","shell.execute_reply":"2023-07-28T07:40:59.78456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# modl.save_weights(\"action.h5\")\nmodl=model = Transformer(\n    num_hid=200,\n    num_head=4,\n    num_feed_forward=400,\n    source_maxlen = FRAME_LEN,\n    target_maxlen=64,\n    num_layers_enc=2,\n    num_layers_dec=1,\n    num_classes=62\n)\nmodl.load_weights('action.h5')","metadata":{"execution":{"iopub.status.busy":"2023-07-28T06:53:22.09195Z","iopub.execute_input":"2023-07-28T06:53:22.092406Z","iopub.status.idle":"2023-07-28T06:53:22.259982Z","shell.execute_reply.started":"2023-07-28T06:53:22.092374Z","shell.execute_reply":"2023-07-28T06:53:22.258072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"  # Replace with the correct path to your saved model.\n\n# Evaluate the model on the training dataset\ntrain_loss, train_accuracy = model.evaluate(train_ds, verbose=1)\nprint(f\"Training Loss: {train_loss}, Training Accuracy: {train_accuracy}\")\n\n# Evaluate the model on the validation dataset\nvalid_loss, valid_accuracy = model.evaluate(valid_ds, verbose=1)\nprint(f\"Validation Loss: {valid_loss}, Validation Accuracy: {valid_accuracy}\")\n\n# Evaluate the model on the test dataset\ntest_loss, test_accuracy = model.evaluate(test_ds, verbose=1)\nprint(f\"Test Loss: {test_loss}, Test Accuracy: {test_accuracy}\")","metadata":{"execution":{"iopub.status.busy":"2023-07-28T09:57:41.498917Z","iopub.execute_input":"2023-07-28T09:57:41.500141Z","iopub.status.idle":"2023-07-28T09:58:07.152584Z","shell.execute_reply.started":"2023-07-28T09:57:41.500082Z","shell.execute_reply":"2023-07-28T09:58:07.151433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction=model.generate(next(iter(test_ds))[0],tf.constant(60, dtype=tf.int32))\n","metadata":{"execution":{"iopub.status.busy":"2023-07-28T08:00:47.324432Z","iopub.execute_input":"2023-07-28T08:00:47.324855Z","iopub.status.idle":"2023-07-28T08:00:49.094815Z","shell.execute_reply.started":"2023-07-28T08:00:47.324824Z","shell.execute_reply":"2023-07-28T08:00:49.093775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds=prediction.numpy()\ntarget=next(iter(test_ds))[1].numpy() \nfor i in range(tf.shape(next(iter(test_ds))[0])[0]):\n            target_text = \"\".join([idx_to_char[_] for _ in target[i, :]])\n            prediction = \"\"\n            for idx in preds[i, :]:\n                prediction += idx_to_char[idx]\n                if idx == 61:\n                    break\n            print(f\"target:     {target_text.replace('-','')}\")\n            print(f\"prediction: {prediction}\\n\")","metadata":{"execution":{"iopub.status.busy":"2023-07-28T08:00:49.99275Z","iopub.execute_input":"2023-07-28T08:00:49.993791Z","iopub.status.idle":"2023-07-28T08:00:50.015503Z","shell.execute_reply.started":"2023-07-28T08:00:49.993733Z","shell.execute_reply":"2023-07-28T08:00:50.014352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"preds=predictions.numpy()\nfor i in range(tf.shape(next(iter(test_ds))[0])[0]):\n            target_text = \"\".join([idx_to_char[_] for _ in target[i, :]])\n            prediction = \"\"\n            for idx in preds[i, :]:\n                prediction += idx_to_char[idx]\n                if idx == 61:\n                    break\n            print(f\"target:     {target_text.replace('-','')}\")\n            print(f\"prediction: {prediction}\\n\")","metadata":{}},{"cell_type":"code","source":"next(iter(test_ds))[0]","metadata":{"execution":{"iopub.status.busy":"2023-07-28T07:23:51.724636Z","iopub.execute_input":"2023-07-28T07:23:51.725038Z","iopub.status.idle":"2023-07-28T07:23:51.739457Z","shell.execute_reply.started":"2023-07-28T07:23:51.725005Z","shell.execute_reply":"2023-07-28T07:23:51.73814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(FEATURE_COLUMNS))","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:14:22.724031Z","iopub.execute_input":"2023-07-28T10:14:22.724473Z","iopub.status.idle":"2023-07-28T10:14:22.730199Z","shell.execute_reply.started":"2023-07-28T10:14:22.724437Z","shell.execute_reply":"2023-07-28T10:14:22.729155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}