{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What I mainly do (see Version 18 for best score)\n\n1. Change FRAME_LEN from 128 to 256\n2. Add more convolutions and a skip connection into LandmarkEmbedding\n\n\n# About Dataset\n\nCurrently I don't plan to make my dataset public, but this notebook uses the same dataset as ROHITH INGILELA's [work](https://www.kaggle.com/code/irohith/aslfr-preprocess-dataset)","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nimport json\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-06-15T04:04:34.998749Z","iopub.execute_input":"2023-06-15T04:04:34.999088Z","iopub.status.idle":"2023-06-15T04:04:44.603077Z","shell.execute_reply.started":"2023-06-15T04:04:34.999063Z","shell.execute_reply":"2023-06-15T04:04:44.601986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open (\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\", \"r\") as f:\n    char_to_num = json.load(f)\n\nnum_to_char = {j:i for i,j in char_to_num.items()}\n\ninpdir = \"/kaggle/input/asl-fingerspelling\"\ndf = pd.read_csv(f'{inpdir}/train.csv')\n\nLIP = [\n    61, 185, 40, 39, 37, 267, 269, 270, 409,\n    291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n    78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n    95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n]\n\nFACE = [f'x_face_{i}' for i in LIP] + [f'y_face_{i}' for i in LIP] + [f'z_face_{i}' for i in LIP]\nLHAND = [f'x_left_hand_{i}' for i in range(21)] + [f'y_left_hand_{i}' for i in range(21)] + [f'z_left_hand_{i}' for i in range(21)]\nRHAND = [f'x_right_hand_{i}' for i in range(21)] + [f'y_right_hand_{i}' for i in range(21)] + [f'z_right_hand_{i}' for i in range(21)]\nPOSE = [f'x_pose_{i}' for i in range(33)] + [f'y_pose_{i}' for i in range(33)] + [f'z_pose_{i}' for i in range(33)]\n\nSEL_COLS = FACE + LHAND + RHAND + POSE\nFRAME_LEN = 256","metadata":{"execution":{"iopub.status.busy":"2023-06-15T04:04:44.605187Z","iopub.execute_input":"2023-06-15T04:04:44.605911Z","iopub.status.idle":"2023-06-15T04:04:44.754633Z","shell.execute_reply.started":"2023-06-15T04:04:44.60588Z","shell.execute_reply":"2023-06-15T04:04:44.753087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RHAND_IDX = [i for i in range(len(SEL_COLS)) if 'right_hand' in SEL_COLS[i]]\nLHAND_IDX = [i for i in range(len(SEL_COLS)) if 'left_hand' in SEL_COLS[i]]\nFACE_IDX = [i for i in range(len(SEL_COLS)) if 'face' in SEL_COLS[i]]\nRPOSE_IDX = [i for i, col in enumerate(SEL_COLS)  if  \"pose\" in col and col[-2:].isnumeric() and 22>=int(col[-2:])>=13 and int(col[-2:])%2==0]\nLPOSE_IDX = [i for i, col in enumerate(SEL_COLS)  if  \"pose\" in col and col[-2:].isnumeric() and 22>=int(col[-2:])>=13 and int(col[-2:])%2==1]\n\n\ndef resize_pad(x):\n    if tf.shape(x)[0] < FRAME_LEN:\n        x = tf.pad(x, ([[0, FRAME_LEN-tf.shape(x)[0]], [0, 0], [0, 0]]))\n    else:\n        x = tf.image.resize(x, (FRAME_LEN, tf.shape(x)[1]))\n    return x\n\ndef preprocess_(x):\n    rhand = tf.gather(x, RHAND_IDX, axis=1)\n    lhand = tf.gather(x, LHAND_IDX, axis=1)\n    rpose = tf.gather(x, RPOSE_IDX, axis=1)\n    lpose = tf.gather(x, LPOSE_IDX, axis=1)\n\n    rnan_idx = tf.reduce_any(tf.math.is_nan(rhand), axis=1)\n    lnan_idx = tf.reduce_any(tf.math.is_nan(lhand), axis=1)\n\n    rnans = tf.math.count_nonzero(rnan_idx)\n    lnans = tf.math.count_nonzero(lnan_idx)\n\n    # For dominant hand\n    if rnans > lnans:\n        hand = lhand\n        pose = lpose\n\n        hand_x = hand[:, 0*(len(LHAND_IDX)//3) : 1*(len(LHAND_IDX)//3)]\n        hand_y = hand[:, 1*(len(LHAND_IDX)//3) : 2*(len(LHAND_IDX)//3)]\n        hand_z = hand[:, 2*(len(LHAND_IDX)//3) : 3*(len(LHAND_IDX)//3)]\n        hand = tf.concat([1-hand_x, hand_y, hand_z], axis=1)\n\n        pose_x = pose[:, 0*(len(LPOSE_IDX)//3) : 1*(len(LPOSE_IDX)//3)]\n        pose_y = pose[:, 1*(len(LPOSE_IDX)//3) : 2*(len(LPOSE_IDX)//3)]\n        pose_z = pose[:, 2*(len(LPOSE_IDX)//3) : 3*(len(LPOSE_IDX)//3)]\n        pose = tf.concat([1-pose_x, pose_y, pose_z], axis=1)\n\n    else:\n        hand = rhand\n        pose = rpose\n\n    hand_x = hand[:, 0*(len(LHAND_IDX)//3) : 1*(len(LHAND_IDX)//3)]\n    hand_y = hand[:, 1*(len(LHAND_IDX)//3) : 2*(len(LHAND_IDX)//3)]\n    hand_z = hand[:, 2*(len(LHAND_IDX)//3) : 3*(len(LHAND_IDX)//3)]\n    hand = tf.concat([hand_x[..., tf.newaxis], hand_y[..., tf.newaxis], hand_z[..., tf.newaxis]], axis=-1)\n\n\n    # hand = tf.concat([rhand,lhand],axis=1)\n    # HAND_IDX = RHAND_IDX + LHAND_IDX\n    # hand_x = hand[:, 0*(len(HAND_IDX)//3) : 1*(len(HAND_IDX)//3)]\n    # hand_y = hand[:, 1*(len(HAND_IDX)//3) : 2*(len(HAND_IDX)//3)]\n    # hand_z = hand[:, 2*(len(HAND_IDX)//3) : 3*(len(HAND_IDX)//3)]\n    # hand = tf.concat([hand_x[..., tf.newaxis], hand_y[..., tf.newaxis], hand_z[..., tf.newaxis]], axis=-1)\n\n    mean = tf.math.reduce_mean(hand, axis=1)[:, tf.newaxis, :]\n    std = tf.math.reduce_std(hand, axis=1)[:, tf.newaxis, :]\n    hand = (hand - mean) / std\n\n    pose_x = pose[:, 0*(len(LPOSE_IDX)//3) : 1*(len(LPOSE_IDX)//3)]\n    pose_y = pose[:, 1*(len(LPOSE_IDX)//3) : 2*(len(LPOSE_IDX)//3)]\n    pose_z = pose[:, 2*(len(LPOSE_IDX)//3) : 3*(len(LPOSE_IDX)//3)]\n    pose = tf.concat([pose_x[..., tf.newaxis], pose_y[..., tf.newaxis], pose_z[..., tf.newaxis]], axis=-1)\n\n    x = tf.concat([hand, pose], axis=1)\n    # x = hand\n    x = resize_pad(x)\n\n    x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n    x = tf.reshape(x, (FRAME_LEN, len(LHAND_IDX)+len(LPOSE_IDX)))\n    # diff_x = x[1:] - x[:-1]\n    # diff_x = tf.concat([tf.zeros_like(diff_x[:1]), diff_x], axis=0)\n    # x = tf.concat([x, diff_x], axis=1)\n    return x\n\ndef preprocess_fn(x,phrase):\n    x = preprocess_(x)\n    return x,phrase\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-15T04:04:44.75702Z","iopub.execute_input":"2023-06-15T04:04:44.757412Z","iopub.status.idle":"2023-06-15T04:04:44.779338Z","shell.execute_reply.started":"2023-06-15T04:04:44.757366Z","shell.execute_reply":"2023-06-15T04:04:44.777754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntable = tf.lookup.StaticHashTable(\n    initializer=tf.lookup.KeyValueTensorInitializer(\n        keys=list(char_to_num.keys()),\n        values=list(char_to_num.values()),\n    ),\n    default_value=tf.constant(-1),\n    name=\"class_weight\"\n)\n\nmask_idx = char_to_num['#']\ndef decode_fn(record_bytes):\n    schema = {COL: tf.io.VarLenFeature(dtype=tf.float32) for COL in SEL_COLS}\n    schema[\"phrase\"] = tf.io.FixedLenFeature([], dtype=tf.string)\n    features = tf.io.parse_single_example(record_bytes, schema)\n    phrase = features[\"phrase\"]\n    landmarks = ([tf.sparse.to_dense(features[COL]) for COL in SEL_COLS])\n    landmarks = tf.transpose(landmarks)\n\n#     mask = tf.math.less(landmarks, -2)\n# #     nan_tensor = tf.fill(tf.shape(landmarks), tf.constant(np.nan, dtype=tf.float32))\n#     nan_tensor = tf.fill(tf.shape(landmarks), tf.constant(0, dtype=tf.float32))\n#     landmarks = tf.where(mask, nan_tensor, landmarks)\n    \n    phrase = ';' + phrase + '['\n    phrase = tf.strings.bytes_split(phrase)\n    phrase = table.lookup(phrase)\n    phrase = tf.pad(phrase, paddings=[[0, 64 - tf.shape(phrase)[0]]], constant_values=mask_idx)\n    return landmarks, phrase\n    \n    return landmarks, phrase\n\ninpdir = \"/kaggle/input/k/nightsh4de/aslfr-preprocess-dataset\"\ntffiles = df.file_id.map(lambda x: f'{inpdir}/tfds/{x}.tfrecord').unique()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-15T04:04:44.782508Z","iopub.execute_input":"2023-06-15T04:04:44.782904Z","iopub.status.idle":"2023-06-15T04:04:44.943824Z","shell.execute_reply.started":"2023-06-15T04:04:44.78287Z","shell.execute_reply":"2023-06-15T04:04:44.942579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 32\nval_len = 1 #int(0.2 * len(tffiles))\nval_batch_size = 1\ntrain_dataset = tf.data.TFRecordDataset(tffiles[val_len:]).map(decode_fn).map(preprocess_fn).batch(batch_size).prefetch(buffer_size=tf.data.AUTOTUNE)\nval_dataset = tf.data.TFRecordDataset(tffiles[:val_len]).map(decode_fn).map(preprocess_fn).batch(val_batch_size).prefetch(buffer_size=tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T04:04:44.945426Z","iopub.execute_input":"2023-06-15T04:04:44.94582Z","iopub.status.idle":"2023-06-15T04:04:47.1273Z","shell.execute_reply.started":"2023-06-15T04:04:44.945787Z","shell.execute_reply":"2023-06-15T04:04:47.125851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TokenEmbedding(layers.Layer):\n    def __init__(self, num_vocab=1000, maxlen=100, num_hid=64):\n        super().__init__()\n        self.emb = tf.keras.layers.Embedding(num_vocab, num_hid)\n        self.pos_emb = layers.Embedding(input_dim=maxlen, output_dim=num_hid)\n\n    def call(self, x):\n        maxlen = tf.shape(x)[-1]\n        x = self.emb(x)\n        positions = tf.range(start=0, limit=maxlen, delta=1)\n        positions = self.pos_emb(positions)\n        return x + positions\n\n\nclass LandmarkEmbedding(layers.Layer):\n    def __init__(self, num_hid=64, maxlen=100):\n        # add two stem conv and one identity shortcut\n        super().__init__()\n        init_dim = 342\n        K_SIZE = 17\n        \n        self.stem1 = tf.keras.layers.Conv1D(\n            init_dim//2, 1, padding=\"same\", activation=\"relu\"\n        )\n        self.stem1_dc = tf.keras.layers.DepthwiseConv1D(17, activation=\"relu\", padding='same')\n\n        self.stem2 = tf.keras.layers.Conv1D(\n            num_hid, 1, padding=\"same\", activation=\"relu\"\n        )\n        self.stem2_dc = tf.keras.layers.DepthwiseConv1D(17, activation=\"relu\", padding='same')\n\n        self.conv1 = tf.keras.layers.Conv1D(\n            num_hid, 1, padding=\"same\", activation=\"relu\"\n        )\n        self.conv1_dc = tf.keras.layers.DepthwiseConv1D(17, activation=\"relu\", padding='same')\n        \n        self.conv2 = tf.keras.layers.Conv1D(\n            num_hid, 1, padding=\"same\", activation=\"relu\"\n        )\n        self.conv2_dc = tf.keras.layers.DepthwiseConv1D(17, activation=\"relu\", padding='same')\n\n        self.conv3 = tf.keras.layers.Conv1D(\n            num_hid, 1, padding=\"same\", activation=\"relu\"\n        )\n        self.conv3_dc = tf.keras.layers.DepthwiseConv1D(17, activation=\"relu\", padding='same')\n\n        self.bn1 = tf.keras.layers.BatchNormalization()\n        self.bn2 = tf.keras.layers.BatchNormalization()\n\n        self.layernorm = layers.LayerNormalization(epsilon=1e-6)\n    def call(self, x):\n        #x = x[..., tf.newaxis]\n        x = self.stem1(x)\n        x = self.stem1_dc(x)\n        x = self.bn1(x)\n        x = self.stem2(x)\n        x = self.stem2_dc(x)\n        x = self.bn2(x)\n        identity = x\n        x = self.conv1(x)\n        x = self.conv1_dc(x)\n        x = self.conv2(x)\n        x = self.conv2_dc(x)\n        x = self.conv3(x)\n        x = self.conv3_dc(x)\n        x = self.layernorm(x+identity)\n        return x\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-06-15T04:04:47.12882Z","iopub.execute_input":"2023-06-15T04:04:47.129134Z","iopub.status.idle":"2023-06-15T04:04:47.143413Z","shell.execute_reply.started":"2023-06-15T04:04:47.12911Z","shell.execute_reply":"2023-06-15T04:04:47.142714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TransformerEncoder(layers.Layer):\n    def __init__(self, embed_dim, num_heads, feed_forward_dim, rate=0.1):\n        super().__init__()\n        self.att = layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        self.ffn = keras.Sequential(\n            [\n                layers.Dense(feed_forward_dim, activation=\"relu\"),\n                layers.Dense(embed_dim),\n            ]\n        )\n        self.layernorm1 = layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm2 = layers.LayerNormalization(epsilon=1e-6)\n        self.dropout1 = layers.Dropout(rate)\n        self.dropout2 = layers.Dropout(rate)\n\n    def call(self, inputs, training):\n        attn_output = self.att(inputs, inputs)\n        attn_output = self.dropout1(attn_output, training=training)\n        out1 = self.layernorm1(inputs + attn_output)\n        ffn_output = self.ffn(out1)\n        ffn_output = self.dropout2(ffn_output, training=training)\n        return self.layernorm2(out1 + ffn_output)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T04:04:47.144376Z","iopub.execute_input":"2023-06-15T04:04:47.144899Z","iopub.status.idle":"2023-06-15T04:04:47.163942Z","shell.execute_reply.started":"2023-06-15T04:04:47.144842Z","shell.execute_reply":"2023-06-15T04:04:47.162655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TransformerDecoder(layers.Layer):\n    def __init__(self, embed_dim, num_heads, feed_forward_dim, dropout_rate=0.1):\n        super().__init__()\n        self.layernorm1 = layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm2 = layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm3 = layers.LayerNormalization(epsilon=1e-6)\n        self.self_att = layers.MultiHeadAttention(\n            num_heads=num_heads, key_dim=embed_dim\n        )\n        self.enc_att = layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        self.self_dropout = layers.Dropout(0.5)\n        self.enc_dropout = layers.Dropout(0.1)\n        self.ffn_dropout = layers.Dropout(0.1)\n        self.ffn = keras.Sequential(\n            [\n                layers.Dense(feed_forward_dim, activation=\"relu\"),\n                layers.Dense(embed_dim),\n            ]\n        )\n\n    def causal_attention_mask(self, batch_size, n_dest, n_src, dtype):\n        \"\"\"Masks the upper half of the dot product matrix in self attention.\n\n        This prevents flow of information from future tokens to current token.\n        1's in the lower triangle, counting from the lower right corner.\n        \"\"\"\n        i = tf.range(n_dest)[:, None]\n        j = tf.range(n_src)\n        m = i >= j - n_src + n_dest\n        mask = tf.cast(m, dtype)\n        mask = tf.reshape(mask, [1, n_dest, n_src])\n        mult = tf.concat(\n            #[tf.expand_dims(batch_size, -1), tf.constant([1, 1], dtype=tf.int32)], 0\n            [batch_size[..., tf.newaxis], tf.constant([1, 1], dtype=tf.int32)], 0\n        )\n        return tf.tile(mask, mult)\n\n    def call(self, enc_out, target):\n        input_shape = tf.shape(target)\n        batch_size = input_shape[0]\n        seq_len = input_shape[1]\n        causal_mask = self.causal_attention_mask(batch_size, seq_len, seq_len, tf.bool)\n        target_att = self.self_att(target, target, attention_mask=causal_mask)\n        target_norm = self.layernorm1(target + self.self_dropout(target_att))\n        enc_out = self.enc_att(target_norm, enc_out)\n        enc_out_norm = self.layernorm2(self.enc_dropout(enc_out) + target_norm)\n        ffn_out = self.ffn(enc_out_norm)\n        ffn_out_norm = self.layernorm3(enc_out_norm + self.ffn_dropout(ffn_out))\n        return ffn_out_norm","metadata":{"execution":{"iopub.status.busy":"2023-06-15T04:04:47.165167Z","iopub.execute_input":"2023-06-15T04:04:47.166516Z","iopub.status.idle":"2023-06-15T04:04:47.181267Z","shell.execute_reply.started":"2023-06-15T04:04:47.16644Z","shell.execute_reply":"2023-06-15T04:04:47.180087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Transformer(keras.Model):\n    def __init__(\n        self,\n        num_hid=64,\n        num_head=2,\n        num_feed_forward=128,\n        source_maxlen=100,\n        target_maxlen=100,\n        num_layers_enc=4,\n        num_layers_dec=1,\n        num_classes=10,\n    ):\n        super().__init__()\n        self.loss_metric = keras.metrics.Mean(name=\"loss\")\n        self.num_layers_enc = num_layers_enc\n        self.num_layers_dec = num_layers_dec\n        self.target_maxlen = target_maxlen\n        self.num_classes = num_classes\n\n        self.enc_input = LandmarkEmbedding(num_hid=num_hid, maxlen=source_maxlen)\n        self.dec_input = TokenEmbedding(\n            num_vocab=num_classes, maxlen=target_maxlen, num_hid=num_hid\n        )\n\n        self.encoder = keras.Sequential(\n            [self.enc_input]\n            + [\n                TransformerEncoder(num_hid, num_head, num_feed_forward)\n                for _ in range(num_layers_enc)\n            ]\n        )\n\n        for i in range(num_layers_dec):\n            setattr(\n                self,\n                f\"dec_layer_{i}\",\n                TransformerDecoder(num_hid, num_head, num_feed_forward),\n            )\n\n        self.classifier = layers.Dense(num_classes)\n\n    def decode(self, enc_out, target):\n        y = self.dec_input(target)\n        for i in range(self.num_layers_dec):\n            y = getattr(self, f\"dec_layer_{i}\")(enc_out, y)\n        return y\n\n    def call(self, inputs):\n        source = inputs[0]\n        target = inputs[1]\n        x = self.encoder(source)\n        y = self.decode(x, target)\n        return self.classifier(y)\n\n    @property\n    def metrics(self):\n        return [self.loss_metric]\n\n    def train_step(self, batch):\n        \"\"\"Processes one batch inside model.fit().\"\"\"\n        source = batch[0]\n        target = batch[1]\n        dec_input = target[:, :-1]\n        dec_target = target[:, 1:]\n        with tf.GradientTape() as tape:\n            preds = self([source, dec_input])\n            one_hot = tf.one_hot(dec_target, depth=self.num_classes)\n            mask = tf.math.logical_not(tf.math.equal(dec_target, mask_idx))\n            loss = self.compiled_loss(one_hot, preds, sample_weight=mask)\n        trainable_vars = self.trainable_variables\n        gradients = tape.gradient(loss, trainable_vars)\n        self.optimizer.apply_gradients(zip(gradients, trainable_vars))\n        self.loss_metric.update_state(loss)\n        return {\"loss\": self.loss_metric.result()}\n\n    def test_step(self, batch):\n        source = batch[0]\n        target = batch[1]\n        dec_input = target[:, :-1]\n        dec_target = target[:, 1:]\n        preds = self([source, dec_input])\n        one_hot = tf.one_hot(dec_target, depth=self.num_classes)\n        mask = tf.math.logical_not(tf.math.equal(dec_target, mask_idx))\n        loss = self.compiled_loss(one_hot, preds, sample_weight=mask)\n        self.loss_metric.update_state(loss)\n        return {\"loss\": self.loss_metric.result()}\n\n    def generate(self, source, target_start_token_idx):\n        \"\"\"Performs inference over one batch of inputs using greedy decoding.\"\"\"\n        bs = tf.shape(source)[0]\n        enc = self.encoder(source)\n        dec_input = tf.ones((bs, 1), dtype=tf.int32) * target_start_token_idx\n        dec_logits = []\n        for i in range(self.target_maxlen - 1):\n            dec_out = self.decode(enc, dec_input)\n            logits = self.classifier(dec_out)\n            logits = tf.argmax(logits, axis=-1, output_type=tf.int32)\n            #last_logit = tf.expand_dims(logits[:, -1], axis=-1)\n            last_logit = logits[:, -1][..., tf.newaxis]\n            dec_logits.append(last_logit)\n            dec_input = tf.concat([dec_input, last_logit], axis=-1)\n        return dec_input","metadata":{"execution":{"iopub.status.busy":"2023-06-15T04:04:47.184633Z","iopub.execute_input":"2023-06-15T04:04:47.185029Z","iopub.status.idle":"2023-06-15T04:04:47.205156Z","shell.execute_reply.started":"2023-06-15T04:04:47.184992Z","shell.execute_reply":"2023-06-15T04:04:47.204338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DisplayOutputs(keras.callbacks.Callback):\n    def __init__(\n        self, batch, idx_to_token, target_start_token_idx=27, target_end_token_idx=28\n    ):\n        \"\"\"Displays a batch of outputs after every epoch\n\n        Args:\n            batch: A test batch containing the keys \"source\" and \"target\"\n            idx_to_token: A List containing the vocabulary tokens corresponding to their indices\n            target_start_token_idx: A start token index in the target vocabulary\n            target_end_token_idx: An end token index in the target vocabulary\n        \"\"\"\n        self.batch = batch\n        self.target_start_token_idx = target_start_token_idx\n        self.target_end_token_idx = target_end_token_idx\n        self.idx_to_char = idx_to_token\n\n    def on_epoch_end(self, epoch, logs=None):\n#         if epoch % 5 != 0:\n#             return\n        source = self.batch[0]\n        target = self.batch[1].numpy()\n        bs = tf.shape(source)[0]\n        preds = self.model.generate(source, self.target_start_token_idx)\n        preds = preds.numpy()\n        for i in range(bs):\n            target_text = \"\".join([self.idx_to_char[_] for _ in target[i, :]])\n            prediction = \"\"\n            for idx in preds[i, :]:\n                prediction += self.idx_to_char[idx]\n                if idx == self.target_end_token_idx:\n                    break\n            print(f\"target:     {target_text.replace('-','')}\")\n            print(f\"prediction: {prediction}\\n\")","metadata":{"execution":{"iopub.status.busy":"2023-06-15T04:04:47.208346Z","iopub.execute_input":"2023-06-15T04:04:47.208804Z","iopub.status.idle":"2023-06-15T04:04:47.229289Z","shell.execute_reply.started":"2023-06-15T04:04:47.208769Z","shell.execute_reply":"2023-06-15T04:04:47.227948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch = next(iter(val_dataset))\n\nidx_to_char = list(char_to_num.keys())\ntarget_start_token_idx = char_to_num[';']\ntarget_end_token_idx = char_to_num['[']\ndisplay_cb = DisplayOutputs(\n    batch, num_to_char, target_start_token_idx=target_start_token_idx, target_end_token_idx=target_end_token_idx\n)  # set the arguments as per vocabulary index for '<' and '>'\n\nmodel = Transformer(\n    num_hid=200,\n    num_head=2,\n    num_feed_forward=400,\n    target_maxlen=64,\n    num_layers_enc=4,\n    num_layers_dec=2,\n    num_classes=59,\n)\n\nloss_fn = tf.keras.losses.CategoricalCrossentropy(from_logits=True, label_smoothing=0.1,)\noptimizer = keras.optimizers.Adam(0.0001)\nmodel.compile(optimizer=optimizer, loss=loss_fn)\ntf.debugging.set_log_device_placement(True)\nhistory = model.fit(train_dataset, validation_data=val_dataset, callbacks=[display_cb], epochs=30)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T04:04:47.230903Z","iopub.execute_input":"2023-06-15T04:04:47.231849Z","iopub.status.idle":"2023-06-15T04:06:32.268939Z","shell.execute_reply.started":"2023-06-15T04:04:47.231789Z","shell.execute_reply":"2023-06-15T04:06:32.266928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T04:06:32.270011Z","iopub.status.idle":"2023-06-15T04:06:32.270486Z","shell.execute_reply.started":"2023-06-15T04:06:32.270247Z","shell.execute_reply":"2023-06-15T04:06:32.270264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class PreprocessLayer(tf.keras.layers.Layer):\n#     def __init__(self):\n#         super(PreprocessLayer, self).__init__()\n\n#     def __call__(self, x):\n#         #x = tf.expand_dims(x, 0)\n#         x = x[None]\n#         x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n#         x = x[0]\n#         x = preprocess_(x)\n#         return x\n\n# class TFLiteModel(tf.Module):\n#     def __init__(self, model):\n#         super().__init__()\n#         # Load the feature generation and main models\n#         self.preprocess_layer = PreprocessLayer()\n#         self.model = model\n\n#     @tf.function(input_signature=[tf.TensorSpec(shape=[None, len(SEL_COLS)], dtype=tf.float32, name='inputs')])\n#     def __call__(self, inputs, training=False):\n#         # Preprocess Data\n#         x = tf.cast(inputs, tf.float32)\n#         x = self.preprocess_layer(inputs)\n#         #x = tf.expand_dims(x, 0)\n#         x = x[None]\n#         target_start_token_idx = char_to_num[';']\n#         target_end_token_idx = char_to_num['[']\n#         x = self.model.generate(x, target_start_token_idx)\n#         x = x[0]\n#         idx = tf.argmax(tf.cast(tf.equal(x, target_end_token_idx), tf.int32))\n#         idx = tf.where(tf.math.less(idx, 1), tf.constant(2, dtype=tf.int64), idx)\n#         x = x[1:idx]\n#         x = tf.one_hot(x, 59)\n#         return {'outputs': x}\n    \n# test_dataset = tf.data.TFRecordDataset(tffiles[val_len:]).map(decode_fn).batch(1).prefetch(buffer_size=tf.data.AUTOTUNE)\n# batch = next(iter(test_dataset))\n# pre = PreprocessLayer()\n# print('preprocessed data shape',pre(batch[0][0]).shape)\n# tflitemodel_base = TFLiteModel(model)\n# print('raw data shape ', batch[0][0].shape)\n# print('output shape', tflitemodel_base(batch[0][0])[\"outputs\"].shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T04:06:32.272382Z","iopub.status.idle":"2023-06-15T04:06:32.273243Z","shell.execute_reply.started":"2023-06-15T04:06:32.27309Z","shell.execute_reply":"2023-06-15T04:06:32.273105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TFLiteModel(tf.Module):\n    def __init__(self, model, target_start_token_idx=char_to_num[';'], target_end_token_idx=char_to_num['[']):\n        super(TFLiteModel, self).__init__()\n        self.model = model\n        self.target_start_token_idx = target_start_token_idx\n        self.target_end_token_idx = target_end_token_idx\n    \n    @tf.function(input_signature=[tf.TensorSpec(shape=[None, len(SEL_COLS)], dtype=tf.float32, name='inputs')])\n    def __call__(self, inputs, training=False):\n        # Preprocess Data\n        x = tf.cast(inputs, tf.float32)\n        x = x[None]\n        x = tf.cond(tf.shape(x)[1] == 0, lambda: tf.zeros((1, 1, len(SEL_COLS))), lambda: tf.identity(x))\n        x = x[0]\n        x = preprocess_(x)\n        x = x[None]\n        x = self.model.generate(x, self.target_start_token_idx)\n        x = x[0]\n        idx = tf.argmax(tf.cast(tf.equal(x, self.target_end_token_idx), tf.int32))\n        idx = tf.where(tf.math.less(idx, 1), tf.constant(2, dtype=tf.int64), idx)\n        x = x[1:idx]\n        x = tf.one_hot(x, 59)\n        return {'outputs': x}\n    \ndef load_relevant_data_subset(pq_path):\n    \n    return pd.read_parquet(pq_path, columns=SEL_COLS)\nfile_id = df.file_id.iloc[0]\ninpdir = \"/kaggle/input/asl-fingerspelling\"\npqfile = f\"{inpdir}/train_landmarks/{file_id}.parquet\"\nseq_refs = df.loc[df.file_id == file_id]\nseqs = load_relevant_data_subset(pqfile)\n\nseq_id = seq_refs.sequence_id.iloc[0]\nframes = seqs.iloc[seqs.index == seq_id]\nphrase = str(df.loc[df.sequence_id == seq_id].phrase.iloc[0])\n\nprint(preprocess_(frames).shape)\ntflitemodel_base = TFLiteModel(model)\ntflitemodel_base(frames)[\"outputs\"].shape\nprint(tflitemodel_base(frames)[\"outputs\"])","metadata":{"execution":{"iopub.status.busy":"2023-06-15T04:06:32.274099Z","iopub.status.idle":"2023-06-15T04:06:32.274812Z","shell.execute_reply.started":"2023-06-15T04:06:32.27465Z","shell.execute_reply":"2023-06-15T04:06:32.274666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras_model_converter = tf.lite.TFLiteConverter.from_keras_model(tflitemodel_base)\nkeras_model_converter.target_spec.supported_ops = [tf.lite.OpsSet.TFLITE_BUILTINS]#, tf.lite.OpsSet.SELECT_TF_OPS]\ntflite_model = keras_model_converter.convert()\nwith open('/kaggle/working/model.tflite', 'wb') as f:\n    f.write(tflite_model)\n    \ninfargs = {\"selected_columns\" : SEL_COLS}\n\nwith open('inference_args.json', \"w\") as json_file:\n    json.dump(infargs, json_file)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T04:06:32.2758Z","iopub.status.idle":"2023-06-15T04:06:32.276627Z","shell.execute_reply.started":"2023-06-15T04:06:32.276446Z","shell.execute_reply":"2023-06-15T04:06:32.276481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip submission.zip  './model.tflite' './inference_args.json'","metadata":{"execution":{"iopub.status.busy":"2023-06-15T04:06:32.277646Z","iopub.status.idle":"2023-06-15T04:06:32.278162Z","shell.execute_reply.started":"2023-06-15T04:06:32.277824Z","shell.execute_reply":"2023-06-15T04:06:32.277838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# interpreter = tf.lite.Interpreter(\"model.tflite\")\n\n# REQUIRED_SIGNATURE = \"serving_default\"\n# REQUIRED_OUTPUT = \"outputs\"\n\n# with open (\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\", \"r\") as f:\n#     character_map = json.load(f)\n# rev_character_map = {j:i for i,j in character_map.items()}\n\n# found_signatures = list(interpreter.get_signature_list().keys())\n\n# if REQUIRED_SIGNATURE not in found_signatures:\n#     raise KernelEvalException('Required input signature not found.')\n\n# prediction_fn = interpreter.get_signature_runner(\"serving_default\")\n# output = prediction_fn(inputs=batch[0][0])\n# prediction_str = \"\".join([rev_character_map.get(s, \"\") for s in np.argmax(output[REQUIRED_OUTPUT], axis=1)])\n# print(prediction_str)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T04:06:32.279916Z","iopub.status.idle":"2023-06-15T04:06:32.280905Z","shell.execute_reply.started":"2023-06-15T04:06:32.280628Z","shell.execute_reply":"2023-06-15T04:06:32.28066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}