{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Got my import\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport seaborn as sn\nimport tensorflow as tf\n\n\n\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import train_test_split, GroupShuffleSplit\nfrom leven import levenshtein\n\nimport glob\nimport sys\nimport os\nimport math\nimport gc\nimport sys\nimport sklearn\nimport time\nimport json\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\n# TQDM Progress Bar With Pandas Apply Function\ntqdm.pandas()\n\nprint(f'Tensorflow Version {tf.__version__}')\nprint(f'Python Version: {sys.version}')\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-15T15:26:38.291609Z","iopub.execute_input":"2023-07-15T15:26:38.291994Z","iopub.status.idle":"2023-07-15T15:26:38.303265Z","shell.execute_reply.started":"2023-07-15T15:26:38.291963Z","shell.execute_reply":"2023-07-15T15:26:38.302123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  Doesn't work for me\n# import tensorflow_addons as tfa\n","metadata":{"execution":{"iopub.status.busy":"2023-07-15T15:26:38.305257Z","iopub.execute_input":"2023-07-15T15:26:38.30601Z","iopub.status.idle":"2023-07-15T15:26:38.311481Z","shell.execute_reply.started":"2023-07-15T15:26:38.305978Z","shell.execute_reply":"2023-07-15T15:26:38.3103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read Character to Ordinal Encoding Mapping\nwith open('/kaggle/input/asl-fingerspelling/character_to_prediction_index.json') as json_file:\n    CHAR2ORD = json.load(json_file)\n    \n# Ordinal to Character Mapping\nORD2CHAR = {j:i for i,j in CHAR2ORD.items()}\n    \n# Character to Ordinal Encoding Mapping   \ndisplay(pd.Series(CHAR2ORD).to_frame('Ordinal Encoding').T)","metadata":{"execution":{"iopub.status.busy":"2023-07-15T15:26:38.313504Z","iopub.execute_input":"2023-07-15T15:26:38.314118Z","iopub.status.idle":"2023-07-15T15:26:38.340379Z","shell.execute_reply.started":"2023-07-15T15:26:38.314087Z","shell.execute_reply":"2023-07-15T15:26:38.339448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# If Notebook Is Run By Committing or In Interactive Mode For Development\nIS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\n# Verbose Setting during training\nVERBOSE = 1 if IS_INTERACTIVE else 2\n# Global Random Seed\nSEED = 20 # Yay first change\n# Number of Frames to resize recording to\nN_TARGET_FRAMES = 128\n# Global debug flag, takes subset of train\nDEBUG = False\n\nN_UNIQUE_CHARACTERS0 = len(CHAR2ORD)\nprint(f'Original number of unique characters {N_UNIQUE_CHARACTERS0}')\nPAD_TOKEN = N_UNIQUE_CHARACTERS0 # This will be the position of Pad Token\nSOS_TOKEN = N_UNIQUE_CHARACTERS0 + 1 # This will be the position of the SOS Token\nEOS_TOKEN = N_UNIQUE_CHARACTERS0 + 2 # This will be the position of EOS Toekn\nN_UNIQUE_CHARACTERS = N_UNIQUE_CHARACTERS0 + 3 # Total number of tokens\nprint(f'Total number of unique characters {N_UNIQUE_CHARACTERS}')\n# Whether to use 30% of data for validation\nUSE_VAL = True\n# Batch Size\nBATCH_SIZE = 64\n# Number of Epochs to Train for\nN_EPOCHS = 20\n# Number of Warmup Epochs in Learning Rate Scheduler\nN_WARMUP_EPOCHS = 10\n# Maximum Learning Rate\nLR_MAX = 1e-3\n# Weight Decay Ratio as Ratio of Learning Rate\nWD_RATIO = 0.05 # Regulaization to penalize large weights\n# Length of Phrase + EOS Token\nMAX_PHRASE_LENGTH = 31 + 1\n# Whether to Train The model\nTRAIN_MODEL = True\n# Whether to Load Pretrained Weights\nLOAD_WEIGHTS = False\n# Learning Rate Warmup Method [log,exp]\nWARMUP_METHOD = 'exp'","metadata":{"execution":{"iopub.status.busy":"2023-07-15T15:26:38.341847Z","iopub.execute_input":"2023-07-15T15:26:38.342188Z","iopub.status.idle":"2023-07-15T15:26:38.351854Z","shell.execute_reply.started":"2023-07-15T15:26:38.342159Z","shell.execute_reply":"2023-07-15T15:26:38.350958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# MatplotLib Global Settings\nmpl.rcParams.update(mpl.rcParamsDefault)\nmpl.rcParams['xtick.labelsize'] = 16\nmpl.rcParams['ytick.labelsize'] = 16\nmpl.rcParams['axes.labelsize'] = 18\nmpl.rcParams['axes.titlesize'] = 24","metadata":{"execution":{"iopub.status.busy":"2023-07-15T15:26:38.354143Z","iopub.execute_input":"2023-07-15T15:26:38.355057Z","iopub.status.idle":"2023-07-15T15:26:38.368416Z","shell.execute_reply.started":"2023-07-15T15:26:38.35502Z","shell.execute_reply":"2023-07-15T15:26:38.367432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_parquet('/kaggle/input/asl-fingerspelling/train_landmarks/1019715464.parquet')\ndf.head()\n\n    ","metadata":{"execution":{"iopub.status.busy":"2023-07-15T15:26:38.371828Z","iopub.execute_input":"2023-07-15T15:26:38.372105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby(by=['sequence_id'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_FRAMES = df.loc[df.groupby(by ='sequence_id').size().idxmax()].shape[0]\nprint(f'Most frames for a single sequence we have in the parqet: {MAX_FRAMES}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read Train DataFrame\nif DEBUG:\n    train = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv').head(5000)\nelse:\n    train = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\n    \n# Set Train Indexed By sqeuence_id\ntrain_sequence_id = train.set_index('sequence_id')\n\n# Number Of Train Samples\nN_SAMPLES = len(train)\nprint(f'N_SAMPLES: {N_SAMPLES}')\n\ndisplay(train.info())\ndisplay(train.head())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get complete file path to file\ndef get_file_path(path):\n    return f'/kaggle/input/asl-fingerspelling/{path}'\n\ntrain['file_path'] = train['path'].apply(get_file_path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Unique Parquet Files\nINFERENCE_FILE_PATHS = pd.Series(\n        glob.glob('/kaggle/input/aslfr-preprocessing-dataset/train_landmark_subsets/*')\n    )\n\nprint(f'Found {len(INFERENCE_FILE_PATHS)} Inference Pickle Files')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Train/Validation\nif USE_VAL:\n    # TRAIN\n    X_train = np.load('/kaggle/input/asl-spellnet-dataset/X_train.npy')\n    y_train = np.load('/kaggle/input/asl-spellnet-dataset/y_train.npy')[:,:MAX_PHRASE_LENGTH]\n    N_TRAIN_SAMPLES = len(X_train)\n    # VAL\n    X_val = np.load('/kaggle/input/asl-spellnet-dataset/X_val.npy')\n    y_val = np.load('/kaggle/input/asl-spellnet-dataset/y_val.npy')[:,:MAX_PHRASE_LENGTH]\n    N_VAL_SAMPLES = len(X_val)\n    # Test\n    X_test = np.load('/kaggle/input/asl-spellnet-dataset/X_test.npy')\n    y_test = np.load('/kaggle/input/asl-spellnet-dataset/y_test.npy')[:,:MAX_PHRASE_LENGTH]\n    N_TEST_SAMPLES = len(X_test)\n    # Shapes\n    print(f'X_train shape: {X_train.shape}, X_val shape: {X_val.shape}, X_test shape: {X_test.shape}')\n    print(f'y_train shape: {y_train.shape}, y_val shape: {y_val.shape}, y_test shape: {y_test.shape}')\n\n\n# Train On All Data\nelse:\n    # TRAIN\n    X_train = np.load('/kaggle/input/aslfr-preprocessing-dataset/X.npy')\n    y_train = np.load('/kaggle/input/aslfr-preprocessing-dataset/y.npy')[:,:MAX_PHRASE_LENGTH]\n    N_TRAIN_SAMPLES = len(X_train)\n    print(f'X_train shape: {X_train.shape}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example Batch For Debugging\nN_EXAMPLE_BATCH_SAMPLES = 1024\nN_EXAMPLE_BATCH_SAMPLES_SMALL = 32\n# Example Batch\nX_batch = {\n    'frames': np.copy(X_train[:N_EXAMPLE_BATCH_SAMPLES]),\n    'phrase': np.copy(y_train[:N_EXAMPLE_BATCH_SAMPLES]),\n#     'phrase_type': np.copy(y_phrase_type_train[:N_EXAMPLE_BATCH_SAMPLES]),\n}\ny_batch = np.copy(y_train[:N_EXAMPLE_BATCH_SAMPLES])\n# Small Example Batch\nX_batch_small = {\n    'frames': np.copy(X_train[:N_EXAMPLE_BATCH_SAMPLES_SMALL]),\n    'phrase': np.copy(y_train[:N_EXAMPLE_BATCH_SAMPLES_SMALL]),\n#     'phrase_type': np.copy(y_phrase_type_train[:N_EXAMPLE_BATCH_SAMPLES_SMALL]),\n}\ny_batch_small = np.copy(y_train[:N_EXAMPLE_BATCH_SAMPLES_SMALL])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read First Parquet File\n# example_parquet_df = pd.read_parquet(train['file_path'][0])\nexample_parquet_df = pd.read_parquet(INFERENCE_FILE_PATHS[0])\n\n# Each parquet file contains 1000 recordings\nprint(f'# Unique Recording: {example_parquet_df.index.nunique()}')\n# Display DataFrame layout\ndisplay(example_parquet_df.head())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get indices in original dataframe\ndef get_idxs(df, words_pos, words_neg=[], ret_names=True, idxs_pos=None):\n    \"\"\"\n    Returns column indices, or both column indices and names\n    Input: dataframe, body_name, words/letters to exclude, get names or not, exact positions to get \n    \"\"\"\n    idxs = []\n    names = []\n    for w in words_pos:\n        for col_idx, col in enumerate(example_parquet_df.columns):\n            # Exclude Non Landmark Columns\n            if col in ['frame']:\n                continue\n                \n            col_idx = int(col.split('_')[-1])\n            # Check if column name contains all words\n            if (w in col) and (idxs_pos is None or col_idx in idxs_pos) and all([w not in col for w in words_neg]):\n                idxs.append(col_idx)\n                names.append(col)\n    # Convert to Numpy arrays\n    idxs = np.array(idxs)\n    names = np.array(names)\n    # Returns either both column indices and names\n    if ret_names:\n        return idxs, names\n    # Or only columns indices\n    else:\n        return idxs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lips Landmark Face Ids\n# #AM we could try adding more face landmarks.\nLIPS_LANDMARK_IDXS = np.array([\n        61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n        291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n        78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n        95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n    ])\n\n# Landmark Indices for Left/Right hand without z axis in raw data\nLEFT_HAND_IDXS0, LEFT_HAND_NAMES0 = get_idxs(example_parquet_df, ['left_hand'], ['z'])\nRIGHT_HAND_IDXS0, RIGHT_HAND_NAMES0 = get_idxs(example_parquet_df, ['right_hand'], ['z'])\nLIPS_IDXS0, LIPS_NAMES0 = get_idxs(example_parquet_df, ['face'], ['z'], idxs_pos=LIPS_LANDMARK_IDXS)\nCOLUMNS0 = np.concatenate((LEFT_HAND_NAMES0, RIGHT_HAND_NAMES0, LIPS_NAMES0))\nN_COLS0 = len(COLUMNS0)\n# Only X/Y axes are used\nN_DIMS0 = 2\n\nprint(f'N_COLS0: {N_COLS0}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Landmark Indices in subset of dataframe with only COLUMNS selected\nLEFT_HAND_IDXS = np.argwhere(np.isin(COLUMNS0, LEFT_HAND_NAMES0)).squeeze()\nRIGHT_HAND_IDXS = np.argwhere(np.isin(COLUMNS0, RIGHT_HAND_NAMES0)).squeeze()\nLIPS_IDXS = np.argwhere(np.isin(COLUMNS0, LIPS_NAMES0)).squeeze()\nHAND_IDXS = np.concatenate((LEFT_HAND_IDXS, RIGHT_HAND_IDXS), axis=0)\nN_COLS = N_COLS0\n# Only X/Y axes are used\nN_DIMS = 2\n\nprint(f'N_COLS: {N_COLS}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Indices in processed data by axes with only dominant hand\nHAND_X_IDXS = np.array(\n        [idx for idx, name in enumerate(LEFT_HAND_NAMES0) if 'x' in name]\n    ).squeeze()\nHAND_Y_IDXS = np.array(\n        [idx for idx, name in enumerate(LEFT_HAND_NAMES0) if 'y' in name]\n    ).squeeze()\n# Names in processed data by axes\nHAND_X_NAMES = LEFT_HAND_NAMES0[HAND_X_IDXS]\nHAND_Y_NAMES = LEFT_HAND_NAMES0[HAND_Y_IDXS]\nHAND_X_NAMES.shape, HAND_Y_NAMES.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(COLUMNS0).to_frame('columns').T","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Mean/Standard Deviations of data used for normalizing\nMEANS = np.load('/kaggle/input/aslfr-preprocessing-dataset/MEANS.npy').reshape(-1)\nSTDS = np.load('/kaggle/input/aslfr-preprocessing-dataset/STDS.npy').reshape(-1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tensorflow Preprocessing Layer","metadata":{}},{"cell_type":"code","source":"\"\"\"\n    Tensorflow layer to process data in TFLite\n    Data needs to be processed in the model itself, so we can not use Python\n\"\"\" \nclass PreprocessLayer(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayer, self).__init__()\n        self.normalisation_correction = tf.constant(\n                    # Add 0.50 to x coordinates of left hand (original right hand) and substract 0.50 of right hand (original left hand)\n                     [0.50 if 'x' in name else 0.00 for name in LEFT_HAND_NAMES0],\n                dtype=tf.float32,\n            )\n    \n    @tf.function(\n        input_signature=(tf.TensorSpec(shape=[None,N_COLS0], dtype=tf.float32),),\n    )\n    def call(self, data0, resize=True):\n        # Fill NaN Values With 0\n        data = tf.where(tf.math.is_nan(data0), 0.0, data0)\n        \n        # Hacky\n        data = data[None]\n        \n        # Empty Hand Frame Filtering\n        hands = tf.slice(data, [0,0,0], [-1, -1, 84])\n        hands = tf.abs(hands)\n        mask = tf.reduce_sum(hands, axis=2)\n        mask = tf.not_equal(mask, 0)\n        data = data[mask][None]\n        \n        # Pad Zeros\n        N_FRAMES = len(data[0])\n        if N_FRAMES < N_TARGET_FRAMES:\n            data = tf.concat((\n                data,\n                tf.zeros([1,N_TARGET_FRAMES-N_FRAMES,N_COLS], dtype=tf.float32)\n            ), axis=1)\n        # Downsample\n        data = tf.image.resize(\n            data,\n            [1, N_TARGET_FRAMES],\n            method=tf.image.ResizeMethod.BILINEAR,\n        )\n        \n        # Squeeze Batch Dimension\n        data = tf.squeeze(data, axis=[0])\n        \n        return data\n    \npreprocess_layer = PreprocessLayer()\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function To Test Preprocessing Layer\ndef test_preprocess_layer():\n    demo_sequence_id = example_parquet_df.index.unique()[15]\n    demo_raw_data = example_parquet_df.loc[demo_sequence_id, COLUMNS0]\n    data = preprocess_layer(demo_raw_data)\n\n    print(f'demo_raw_data shape: {demo_raw_data.shape}')\n    print(f'data shape: {data.shape}')\n    \n    return data\n    \nif IS_INTERACTIVE:\n    data = test_preprocess_layer()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_dataset(X, y, batch_size=BATCH_SIZE):\n    sample_idxs = np.arange(len(X))\n    while True:\n        # Get random indices\n        random_sample_idxs = np.random.choice(sample_idxs, batch_size)\n        \n        inputs = {\n            'frames': X[random_sample_idxs],\n            'phrase': y[random_sample_idxs],\n        }\n        outputs = y[random_sample_idxs]\n        \n        yield inputs, outputs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train Dataset\ntrain_dataset = get_train_dataset(X_train, y_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training Steps Per Epoch\nTRAIN_STEPS_PER_EPOCH = math.ceil(N_TRAIN_SAMPLES / BATCH_SIZE)\nprint(f'TRAIN_STEPS_PER_EPOCH: {TRAIN_STEPS_PER_EPOCH}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Calidatioin","metadata":{}},{"cell_type":"code","source":"# Validation Set\ndef get_val_dataset(X, y, batch_size=BATCH_SIZE):\n    offsets = np.arange(0, len(X), batch_size)\n    while True:\n        # Iterate over whole validation set\n        for offset in offsets:\n            inputs = {\n                'frames': X[offset:offset+batch_size],\n                'phrase': y[offset:offset+batch_size],\n            }\n            outputs = y[offset:offset+batch_size]\n\n            yield inputs, outputs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Validation Dataset\nprint(USE_VAL)\nif USE_VAL:\n    val_dataset = get_val_dataset(X_val, y_val)\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if USE_VAL:\n    N_VAL_STEPS_PER_EPOCH = math.ceil(N_VAL_SAMPLES / BATCH_SIZE)\n    print(f'N_VAL_STEPS_PER_EPOCH: {N_VAL_STEPS_PER_EPOCH}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Config","metadata":{}},{"cell_type":"code","source":"\n# Epsilon value for layer normalisation\nLAYER_NORM_EPS = 1e-6\n\n# final embedding and transformer embedding size\nUNITS_ENCODER = 384\nUNITS_DECODER = 256\n\n# Transformer\nNUM_BLOCKS_ENCODER = 3\nNUM_BLOCKS_DECODER = 2\nNUM_HEADS = 4\nMLP_RATIO = 2\n\n# Dropout\nEMBEDDING_DROPOUT = 0.00\nMLP_DROPOUT_RATIO = 0.30\nMHA_DROPOUT_RATIO = 0.20\nCLASSIFIER_DROPOUT_RATIO = 0.10\n\n# Initiailizers\nINIT_HE_UNIFORM = tf.keras.initializers.he_uniform\nINIT_GLOROT_UNIFORM = tf.keras.initializers.glorot_uniform\nINIT_ZEROS = tf.keras.initializers.constant(0.0)\n# Activations\nGELU = tf.keras.activations.gelu\n\n# Learning Rate\nLEARNING_RATE = 1e-4","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Landmarks Embedding ","metadata":{}},{"cell_type":"code","source":"# Embeds a landmark using fully connected layers\nclass LandmarkEmbedding(tf.keras.Model):\n    def __init__(self, units, name):\n        super(LandmarkEmbedding, self).__init__(name=f'{name}_embedding')\n        self.units = units\n        self.supports_masking = True\n        \n    def build(self, input_shape):\n        # Embedding for missing landmark in frame, initizlied with zeros\n        self.empty_embedding = self.add_weight(\n            name=f'{self.name}_empty_embedding',\n            shape=[self.units],\n            initializer=INIT_ZEROS,\n        )\n        # Embedding\n        self.dense = tf.keras.Sequential([\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM, activation=GELU), # Can change activation\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name=f'{self.name}_dense')\n\n    def call(self, x):\n        return tf.where(\n                # Checks whether landmark is missing in frame\n                tf.reduce_sum(x, axis=2, keepdims=True) == 0,\n                # If so, the empty embedding is used\n                self.empty_embedding,\n                # Otherwise the landmark data is embedded\n                self.dense(x),\n            )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Embedding","metadata":{}},{"cell_type":"code","source":"# Creates embedding for each frame\nclass Embedding(tf.keras.Model):\n    def __init__(self):\n        super(Embedding, self).__init__()\n        self.supports_masking = True\n    \n    def build(self, input_shape):\n        # Positional embedding for each frame index\n        self.positional_embedding = tf.Variable(\n            initial_value=tf.zeros([N_TARGET_FRAMES, UNITS_ENCODER], dtype=tf.float32),\n            trainable=True,\n            name='embedding_positional_encoder',\n        )\n        # Embedding layer for Landmarks\n        self.dominant_hand_embedding = LandmarkEmbedding(UNITS_ENCODER, 'dominant_hand')\n\n    def call(self, x, training=False):\n        # Normalize\n        x = tf.where(\n                tf.math.equal(x, 0.0),\n                0.0,\n                (x - MEANS) / STDS,\n            )\n        # Dominant Hand\n        x = self.dominant_hand_embedding(x)\n        # Add Positional Encoding\n        x = x + self.positional_embedding\n        \n        return x","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# MultiHead Attention Block","metadata":{}},{"cell_type":"code","source":"# replaced softmax with softmax layer to support masked softmax\ndef scaled_dot_product(q,k,v, softmax, attention_mask):\n    #calculates Q . K(transpose)\n    qkt = tf.matmul(q,k,transpose_b=True)\n    #caculates scaling factor\n    dk = tf.math.sqrt(tf.cast(q.shape[-1],dtype=tf.float32))\n    scaled_qkt = qkt/dk\n    softmax = softmax(scaled_qkt, mask=attention_mask)\n    z = tf.matmul(softmax,v)\n    #shape: (m,Tx,depth), same shape as q,k,v\n    return z\n\nclass MultiHeadAttention(tf.keras.layers.Layer):\n    def __init__(self,d_model, num_of_heads, dropout, d_out=None):\n        super(MultiHeadAttention,self).__init__()\n        self.d_model = d_model\n        self.num_of_heads = num_of_heads\n        self.depth = d_model//num_of_heads # Can change\n        self.wq = [tf.keras.layers.Dense(self.depth//2, use_bias=False) for i in range(num_of_heads)] # depth//2 isn't common, we can try different numbers\n        self.wk = [tf.keras.layers.Dense(self.depth//2, use_bias=False) for i in range(num_of_heads)]\n        self.wv = [tf.keras.layers.Dense(self.depth//2, use_bias=False) for i in range(num_of_heads)]\n        self.softmax = tf.keras.layers.Softmax()\n        self.do = tf.keras.layers.Dropout(dropout)\n        self.supports_masking = True\n        self.wo = tf.keras.layers.Dense(d_model if d_out is None else d_out, use_bias=False)\n        \n    def call(self, q, k, v, attention_mask=None, training=False):\n        \n        multi_attn = []\n        for i in range(self.num_of_heads):\n            Q = self.wq[i](q)\n            K = self.wk[i](k)\n            V = self.wv[i](v)\n            multi_attn.append(scaled_dot_product(Q,K,V, self.softmax, attention_mask))\n            \n        multi_head = tf.concat(multi_attn, axis=-1)\n        multi_head_attention = self.wo(multi_head)\n        multi_head_attention = self.do(multi_head_attention, training=training)\n        \n        return multi_head_attention","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class EncoderTransformerBlock(tf.keras.layers.Layer):\n    def __init__(self, units, num_heads, mlp_ratio, mha_dropout_ratio, mlp_dropout_ratio, **kwargs):\n        super(EncoderTransformerBlock, self).__init__(**kwargs)\n        self.layer_norm_1 = tf.keras.layers.LayerNormalization(epsilon=LAYER_NORM_EPS)\n        self.mha = MultiHeadAttention(units, num_heads, mha_dropout_ratio)\n        self.layer_norm_2 = tf.keras.layers.LayerNormalization(epsilon=LAYER_NORM_EPS)\n        self.mlp = tf.keras.Sequential([\n            tf.keras.layers.Dense(units * mlp_ratio, activation=GELU, kernel_initializer=INIT_GLOROT_UNIFORM, use_bias=False),\n            tf.keras.layers.Dropout(mlp_dropout_ratio),\n            tf.keras.layers.Dense(units, kernel_initializer=INIT_HE_UNIFORM, use_bias=False),\n        ])\n\n    def call(self, inputs, attention_mask, training=False):\n        x = self.layer_norm_1(inputs + self.mha(inputs, inputs, inputs, attention_mask=attention_mask))\n        x = self.layer_norm_2(x + self.mlp(x))\n        return x\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Encoder","metadata":{}},{"cell_type":"code","source":"class Encoder(tf.keras.Model):\n    def __init__(self, num_blocks):\n        super(Encoder, self).__init__(name='encoder')\n        self.num_blocks = num_blocks\n        self.support_masking = True\n        self.blocks = [EncoderTransformerBlock(UNITS_ENCODER, NUM_HEADS, MLP_RATIO, MHA_DROPOUT_RATIO, MLP_DROPOUT_RATIO) for _ in range(num_blocks)]\n\n        # Optional Projection to Decoder Dimension\n        if UNITS_ENCODER != UNITS_DECODER:\n            self.dense_out = tf.keras.layers.Dense(UNITS_DECODER, kernel_initializer=INIT_GLOROT_UNIFORM, use_bias=False)\n            self.apply_dense_out = True\n        else:\n            self.apply_dense_out = False\n\n    def call(self, x, x_inp, training=False):\n        #Attention mask to ignore missing frames\n        attention_mask = tf.where(tf.math.reduce_sum(x_inp, axis=[2]) == 0.0, 0.0, 1.0)\n        attention_mask = tf.expand_dims(attention_mask, axis=1)\n        attention_mask = tf.repeat(attention_mask, repeats=N_TARGET_FRAMES, axis=1)\n        \n        # Iterate input over transformer blocks\n        for block in self.blocks:\n            x = block(x, attention_mask=attention_mask, training=training)\n            \n        # Optional Projection to Decoder Dimension\n        if self.apply_dense_out:\n            x = self.dense_out(x)\n    \n        return x\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class Encoder(tf.keras.Model):\n#     def __init__(self, num_blocks):\n#         super(Encoder, self).__init__(name='encoder')\n#         self.num_blocks = num_blocks\n#         self.support_masking = True\n    \n#     def build(self, input_shape):\n#         self.ln_1s = []\n#         self.mhas = []\n#         self.ln_2s = []\n#         self.mlps = []\n#         for i in range(self.num_blocks):\n#             # Normalization Layer\n#             self.ln_1s.append(tf.keras.layers.LayerNormalization(epsilon=LAYER_NORM_EPS))\n#             # MultiHeads Layer\n#             self.mhas.append(MultiHeadAttention(UNITS_ENCODER, NUM_HEADS, MHA_DROPOUT_RATIO))\n#             # Normalization Layer \n#             self.ln_2s.append(tf.keras.layers.LayerNormalization(epsilon=LAYER_NORM_EPS))\n#             # Multi Layer Preception\n#             self.mlps.append(tf.keras.Sequential([\n#                 tf.keras.layers.Dense(UNITS_ENCODER * MLP_RATIO, activation=GELU, kernel_initializer=INIT_GLOROT_UNIFORM, use_bias=False),\n#                 tf.keras.layers.Dropout(MLP_DROPOUT_RATIO),\n#                 tf.keras.layers.Dense(UNITS_ENCODER, kernel_initializer=INIT_HE_UNIFORM, use_bias=False), # Can change \n#             ]))\n#                 # Optional Projection to Decoder Dimension\n#             if UNITS_ENCODER != UNITS_DECODER:\n#                 self.dense_out = tf.keras.layers.Dense(UNITS_DECODER, kernel_initializer=INIT_GLOROT_UNIFORM, use_bias=False)\n#                 self.apply_dense_out = True\n#             else:\n#                 self.apply_dense_out = False\n                \n#     def call(self, x, x_inp, training=False):\n#         #Attention mask to ignore missing frames\n#         attention_mask = tf.where(tf.math.reduce_sum(x_inp, axis=[2]) == 0.0, 0.0, 1.0)\n#         attention_mask = tf.expand_dims(attention_mask, axis=1)\n#         attention_mask = tf.repeat(attention_mask, repeats=N_TARGET_FRAMES, axis=1)\n#        # Iterate input over transformer blocks\n#         for ln_1, mha, ln_2, mlp in zip(self.ln_1s, self.mhas, self.ln_2s, self.mlps):\n#             x = ln_1(x + mha(x, x, x, attention_mask=attention_mask))\n#             x = ln_2(x + mlp(x))\n            \n#         # Optional Projection to Decoder Dimension\n#         if self.apply_dense_out:\n#             x = self.dense_out(x)\n    \n#         return x","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Decoder Transformer\n","metadata":{}},{"cell_type":"code","source":"class DecoderTransformerBlock(tf.keras.layers.Layer):\n    def __init__(self, units, num_heads, mlp_ratio, mha_dropout_ratio, mlp_dropout_ratio, **kwargs):\n        super(DecoderTransformerBlock, self).__init__(**kwargs)\n        self.layer_norm_1 = tf.keras.layers.LayerNormalization(epsilon=LAYER_NORM_EPS)\n        self.mha = MultiHeadAttention(units, num_heads, mha_dropout_ratio)\n        self.layer_norm_2 = tf.keras.layers.LayerNormalization(epsilon=LAYER_NORM_EPS)\n        self.mlp = tf.keras.Sequential([\n            tf.keras.layers.Dense(units * mlp_ratio, activation=GELU, kernel_initializer=INIT_GLOROT_UNIFORM, use_bias=False),\n            tf.keras.layers.Dropout(mlp_dropout_ratio),\n            tf.keras.layers.Dense(units, kernel_initializer=INIT_HE_UNIFORM, use_bias=False),\n        ])\n\n    def call(self, inputs, encoder_outputs, attention_mask, training=False):\n        x = self.layer_norm_1(inputs + self.mha(inputs, encoder_outputs, encoder_outputs, attention_mask=attention_mask))\n        x = self.layer_norm_2(x + self.mlp(x))\n        return x\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Decoder","metadata":{}},{"cell_type":"code","source":"self_attention_mask =tf.ones((N_TARGET_FRAMES, N_TARGET_FRAMES))\nself_attention_mask = tf.linalg.band_part(self_attention_mask, -1, 0)\ntf.cast(self_attention_mask, tf.float32)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Decoder(tf.keras.Model):\n    def __init__(self, num_blocks):\n        super(Decoder, self).__init__(name='decoder')\n        self.num_blocks = num_blocks\n        self.supports_masking = True\n\n        # Positional Embedding, initialized with zeros\n        self.positional_embedding = tf.Variable(\n            initial_value=tf.zeros([N_TARGET_FRAMES, UNITS_DECODER], dtype=tf.float32),\n            trainable=True,\n            name='embedding_positional_encoder',\n        )\n\n        # Character Embedding\n        self.char_emb = tf.keras.layers.Embedding(N_UNIQUE_CHARACTERS, UNITS_DECODER, embeddings_initializer=INIT_ZEROS)\n        \n        # Positional Encoder MHA\n        self.pos_emb_mha = MultiHeadAttention(UNITS_DECODER, NUM_HEADS, MHA_DROPOUT_RATIO)\n        self.pos_emb_ln = tf.keras.layers.LayerNormalization(epsilon=LAYER_NORM_EPS)\n\n        # Decoder Blocks\n        self.blocks = [DecoderTransformerBlock(UNITS_DECODER, NUM_HEADS, MLP_RATIO, MHA_DROPOUT_RATIO,MLP_DROPOUT_RATIO) for _ in range(num_blocks)]\n\n    def get_causal_attention_mask(self, B):\n        # My version of the mask AM\n        ones = tf.ones((N_TARGET_FRAMES, N_TARGET_FRAMES))\n        mask = tf.linalg.band_part(ones, 0, -1)  \n        mask = tf.transpose(mask)\n        mask = tf.expand_dims(mask, axis=0)\n        mask = tf.tile(mask, [B, 1, 1])\n        mask = tf.cast(mask, tf.float32)\n        return mask\n\n    def call(self, encoder_outputs, phrase, training=False):\n        # Batch Size\n        B = tf.shape(encoder_outputs)[0]\n        # Cast to INT32\n        phrase = tf.cast(phrase, tf.int32)\n        # Prepend SOS Token\n        phrase = tf.pad(phrase, [[0,0], [1,0]], constant_values=SOS_TOKEN, name='prepend_sos_token')\n        # Pad With PAD Token\n        phrase = tf.pad(phrase, [[0,0], [0,N_TARGET_FRAMES-MAX_PHRASE_LENGTH-1]], constant_values=PAD_TOKEN, name='append_pad_token')\n        # Causal Mask\n        causal_mask = self.get_causal_attention_mask(B)\n        # Positional Embedding\n        x = self.positional_embedding + self.char_emb(phrase)\n        # Causal Attention\n        x = self.pos_emb_ln(x + self.pos_emb_mha(x, x, x, attention_mask=causal_mask))\n        # Iterate input over causal_masktransformer blocks\n        for block in self.blocks:\n            x = block(x, encoder_outputs, causal_mask, training=training)\n        # Slice 31 Characters\n        x = tf.slice(x, [0, 0, 0], [-1, MAX_PHRASE_LENGTH, -1])\n        return x\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Non Pad/SOS/EOS Token Accuracy","metadata":{}},{"cell_type":"markdown","source":"# Sparse Categorical Crossentropy With Label Smoothing","metadata":{}},{"cell_type":"code","source":"# TopK accuracy for multi dimensional output\nclass TopKAccuracy(tf.keras.metrics.Metric):\n    def __init__(self, k, **kwargs):\n        super(TopKAccuracy, self).__init__(name=f'top{k}acc', **kwargs)\n        self.top_k_acc = tf.keras.metrics.SparseTopKCategoricalAccuracy(k=k)\n\n    def update_state(self, y_true, y_pred, sample_weight=None):\n        y_true = tf.reshape(y_true, [-1])\n        y_pred = tf.reshape(y_pred, [-1, N_UNIQUE_CHARACTERS])\n        character_idxs = tf.where(y_true < N_UNIQUE_CHARACTERS0)\n        y_true = tf.gather(y_true, character_idxs, axis=0)\n        y_pred = tf.gather(y_pred, character_idxs, axis=0)\n        self.top_k_acc.update_state(y_true, y_pred)\n\n    def result(self):\n        return self.top_k_acc.result()\n    \n    def reset_state(self):\n        self.top_k_acc.reset_state()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# source:: https://stackoverflow.com/questions/60689185/label-smoothing-for-sparse-categorical-crossentropy\n\nclass CustomSCCEWithLS(tf.keras.losses.Loss):\n    def __init__(self, loss_weights, pad_token=PAD_TOKEN, n_unique_characters=N_UNIQUE_CHARACTERS, label_smoothing=0.25, **kwargs):\n        super().__init__(**kwargs)\n        self.loss_weights = loss_weights\n        self.pad_token = pad_token\n        self.n_unique_characters = n_unique_characters\n        self.label_smoothing = label_smoothing\n\n    def call(self, y_true, y_pred):\n        # Filter Pad Tokens\n        idxs = tf.where(y_true != self.pad_token)\n        y_true = tf.gather_nd(y_true, idxs)\n        y_pred = tf.gather_nd(y_pred, idxs)\n        # One Hot Encode Sparsely Encoded Target Sign\n        y_true = tf.cast(y_true, tf.int32)\n        y_true = tf.one_hot(y_true, self.n_unique_characters, axis=1)\n        # Apply loss_weights\n        y_true = y_true * self.loss_weights\n        # Categorical Crossentropy with native label smoothing support\n        loss = tf.keras.losses.categorical_crossentropy(y_true, y_pred, label_smoothing=self.label_smoothing, from_logits=True)\n        loss = tf.math.reduce_mean(loss)\n        return loss\n\n    def get_config(self):\n        base_config = super().get_config()\n        return {**base_config, \"loss_weights\": self.loss_weights.tolist(), \"pad_token\": self.pad_token, \n                \"n_unique_characters\": self.n_unique_characters, \"label_smoothing\": self.label_smoothing}\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"def get_model():\n    # Inputs\n    frames_inp = tf.keras.layers.Input([N_TARGET_FRAMES, N_COLS], dtype=tf.float32, name='frames')\n    phrase_inp = tf.keras.layers.Input([MAX_PHRASE_LENGTH], dtype=tf.int32, name='phrase')\n    # Frames\n    x = frames_inp\n    \n    # Masking \n    x = tf.keras.layers.Masking(mask_value=0.0, input_shape=(N_TARGET_FRAMES, N_COLS))(x)\n    \n    # Embedding\n    x = Embedding()(x)\n    \n    # Encoder Transformer Blocks\n    x = Encoder(NUM_BLOCKS_ENCODER)(x, frames_inp)\n    \n    # Decoder\n    x = Decoder(NUM_BLOCKS_DECODER)(x, phrase_inp)\n    \n    # Classifier\n    x = tf.keras.Sequential([\n        # Dropout\n        tf.keras.layers.Dropout(CLASSIFIER_DROPOUT_RATIO),\n        # Output Neurons\n        tf.keras.layers.Dense(N_UNIQUE_CHARACTERS, activation=tf.keras.activations.linear, kernel_initializer=INIT_HE_UNIFORM, use_bias=False), # can changes initializer/activation\n    ], name='classifier')(x)\n    \n    outputs = x\n    \n    # Create Tensorflow Model\n    model = tf.keras.models.Model(inputs=[frames_inp, phrase_inp], outputs=outputs)\n    \n    #optimizer\n    optimizer = tf.keras.optimizers.Adam(learning_rate=0.0001) # We should try different optimizers with learning rate scheduler\n    \n    # Create Initial Loss Weights All Set To 1\n    loss_weights_tensor = tf.ones([N_UNIQUE_CHARACTERS], dtype=tf.float32)\n    # Set Loss Weight Of Pad Token To 0\n    loss_weights_tensor = tf.tensor_scatter_nd_update(loss_weights_tensor, [[PAD_TOKEN]], [0])\n\n    # Convert the tensor to a numpy array\n    loss_weights = loss_weights_tensor.numpy()\n\n    # Categorical Crossentropy Loss With Label Smoothing\n#     loss = scce_with_ls\n    loss = CustomSCCEWithLS(loss_weights)\n\n    metrics = [\n        TopKAccuracy(1),\n        TopKAccuracy(5),\n    ]\n    \n    model.compile(\n        loss=loss,\n        optimizer=optimizer,\n        metrics=metrics,\n        loss_weights=loss_weights,\n    )\n    \n    return model","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for k, v in X_batch.items():\n    print(f'{k}: {v.shape}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.backend.clear_session()\n\nmodel = get_model()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(model, show_shapes=True, show_dtype=True, show_layer_names=True, show_layer_activations=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" history = model.fit(\n            x=train_dataset,\n            steps_per_epoch=TRAIN_STEPS_PER_EPOCH,\n            epochs=N_EPOCHS,\n            validation_data=val_dataset,\n            validation_steps=N_VAL_STEPS_PER_EPOCH,\n            verbose = VERBOSE,\n        )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save_weights('model.h5')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save history\ntrain_loss = history.history['loss']\ntrain_top1acc = history.history['top1acc']\ntrain_top5acc = history.history['top5acc']\n\nval_loss = history.history['val_loss']\nval_top1acc = history.history['val_top1acc']\nval_top5acc = history.history['val_top5acc']\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nepochs = range(1, len(val_loss) + 1)\n\n# Plotting Loss\nplt.figure(figsize=(6,6))\nplt.plot(epochs, train_loss, 'b', label='Train Loss')\nplt.plot(epochs, val_loss, 'r', label='Validation Loss')\nplt.title('Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.tight_layout()\n\nplt.show();\n\n# Plotting Accuracy\nplt.figure(figsize=(6,6))\nplt.plot(epochs, train_top1acc, 'b', label='Train Top 1 Accuracy')\nplt.plot(epochs, val_top1acc, 'r', label='Validation Top 1 Accuracy')\nplt.title('Top 1 Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Top 1 Accuracy')\nplt.legend()\nplt.tight_layout()\n\nplt.show();\n\nplt.figure(figsize=(6,6))\nplt.plot(epochs, train_top5acc, 'b', label='Train Top 5 Accuracy')\nplt.plot(epochs, val_top5acc, 'r', label='Validation Top 5 Accuracy')\nplt.title('Top 5 Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Top 5 Accuracy')\nplt.legend()\nplt.tight_layout()\n\nplt.show();\n\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Custom callback to update weight decay with learning rate\nclass WeightDecayCallback(tf.keras.callbacks.Callback):\n    def __init__(self, wd_ratio=WD_RATIO):\n        self.step_counter = 0\n        self.wd_ratio = wd_ratio\n    \n    def on_epoch_begin(self, epoch, logs=None):\n        model.optimizer.weight_decay = model.optimizer.learning_rate * self.wd_ratio\n        print(f'learning rate: {model.optimizer.learning_rate.numpy():.2e}, weight decay: {model.optimizer.weight_decay.numpy():.2e}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lrfn(current_step, num_warmup_steps, lr_max, num_cycles=0.50, num_training_steps=N_EPOCHS):\n    \n    if current_step < num_warmup_steps:\n        if WARMUP_METHOD == 'log':\n            return lr_max * 0.10 ** (num_warmup_steps - current_step)\n        else:\n            return lr_max * 2 ** -(num_warmup_steps - current_step)\n    else:\n        progress = float(current_step - num_warmup_steps) / float(max(1, num_training_steps - num_warmup_steps))\n\n        return max(0.0, 0.5 * (1.0 + math.cos(math.pi * float(num_cycles) * 2.0 * progress))) * lr_max","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction ","metadata":{}},{"cell_type":"code","source":"# Output Predictions to string\ndef outputs2phrase(outputs):\n    if outputs.ndim == 2:\n        outputs = np.argmax(outputs, axis=1)\n    \n    return ''.join([ORD2CHAR.get(s, '') for s in outputs])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@tf.function()\ndef predict_phrase(frames):\n    # Add Batch Dimension\n    frames = tf.expand_dims(frames, axis=0)\n    # Start Phrase\n    phrase = tf.fill([1,MAX_PHRASE_LENGTH], PAD_TOKEN)\n\n    for idx in tf.range(MAX_PHRASE_LENGTH):\n        # Cast phrase to int8\n        phrase = tf.cast(phrase, tf.int8)\n        # Predict Next Token\n        outputs = model({\n            'frames': frames,\n            'phrase': phrase,\n        })\n\n        # Add predicted token to input phrase\n        phrase = tf.cast(phrase, tf.int32)\n        phrase = tf.where(\n            tf.range(MAX_PHRASE_LENGTH) < idx + 1,\n            tf.argmax(outputs, axis=2, output_type=tf.int32),\n            phrase,\n        )\n\n    # Squeeze outputs\n    outputs = tf.squeeze(phrase, axis=0)\n    outputs = tf.one_hot(outputs, N_UNIQUE_CHARACTERS)\n\n    # Return a dictionary with the output tensor\n    return outputs\n\n    # Return a dictionary with the output tensor\n    return outputs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute Levenstein Distances\ndef get_ld_test():\n    LD_TEST= []\n    for idx, (frames, phrase_true) in enumerate(zip(tqdm(X_test, total=X_test.shape[0]), y_test)):\n        # Predict Phrase and Convert to String\n        phrase_pred = predict_phrase(frames).numpy()\n        phrase_pred = outputs2phrase(phrase_pred)\n        # True Phrase Ordinal to String\n        phrase_true = outputs2phrase(phrase_true)\n        # Add Levenstein Distance\n        LD_TEST.append({\n            'phrase_true': phrase_true,\n            'phrase_pred': phrase_pred,\n            'levenshtein_distance': levenshtein(phrase_pred, phrase_true),\n        })\n    \n    # Convert to DataFrame\n    LD_TEST_DF = pd.DataFrame(LD_TEST)\n    \n    return LD_TEST_DF","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LD_TEST_DF = get_ld_test()\n\n# Display Errors\ndisplay(LD_TEST_DF.head(30))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LD_TEST_DF['levenshtein_distance'].describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,6))  \nplt.hist(LD_TEST_DF['levenshtein_distance'], bins=30, color='skyblue', edgecolor='black')\nplt.title('Histogram of Levenshtein Distances')  \nplt.xlabel('Levenshtein Distance')  \nplt.ylabel('Count')  \nplt.grid(axis='y', alpha=0.75)  \nplt.show();","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LD_TEST_DF['levenshtein_distance'].mean(), LD_TEST_DF['levenshtein_distance'].std()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LD_TEST_DF.to_csv('LD_TEST_DF')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}