{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":52950,"databundleVersionId":5973250,"sourceType":"competition"},{"sourceId":11694503,"sourceType":"datasetVersion","datasetId":7340011},{"sourceId":11945614,"sourceType":"datasetVersion","datasetId":7509715}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport shutil\nimport pyarrow.parquet as pq\nimport tensorflow as tf\nimport json\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport random\n\nfrom skimage.transform import resize\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tqdm.notebook import tqdm\nfrom matplotlib import animation, rc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:00.643038Z","iopub.execute_input":"2025-05-25T16:16:00.643394Z","iopub.status.idle":"2025-05-25T16:16:00.649893Z","shell.execute_reply.started":"2025-05-25T16:16:00.643368Z","shell.execute_reply":"2025-05-25T16:16:00.648604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:00.656611Z","iopub.execute_input":"2025-05-25T16:16:00.656981Z","iopub.status.idle":"2025-05-25T16:16:00.669722Z","shell.execute_reply.started":"2025-05-25T16:16:00.656956Z","shell.execute_reply":"2025-05-25T16:16:00.668552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_df = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\ntf_records = dataset_df.file_id.map(lambda x: f'/kaggle/input/pre-data-fsp/new_data/{x}.tfrecord').unique()\nprint(f\"List of {len(tf_records)} TFRecord files.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:00.671496Z","iopub.execute_input":"2025-05-25T16:16:00.672356Z","iopub.status.idle":"2025-05-25T16:16:00.866249Z","shell.execute_reply.started":"2025-05-25T16:16:00.672326Z","shell.execute_reply":"2025-05-25T16:16:00.865305Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LPOSE = [13, 15, 17, 19, 21]\nRPOSE = [14, 16, 18, 20, 22]\nPOSE = LPOSE + RPOSE\nX = [f'x_right_hand_{i}' for i in range(21)] + [f'x_left_hand_{i}' for i in range(21)] + [f'x_pose_{i}' for i in POSE]\nY = [f'y_right_hand_{i}' for i in range(21)] + [f'y_left_hand_{i}' for i in range(21)] + [f'y_pose_{i}' for i in POSE]\nZ = [f'z_right_hand_{i}' for i in range(21)] + [f'z_left_hand_{i}' for i in range(21)] + [f'z_pose_{i}' for i in POSE]\nFEATURE_COLUMNS = X + Y + Z\nRHAND_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"right\" in col]\nLHAND_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"left\" in col]\nRPOSE_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"pose\" in col and int(col[-2:]) in RPOSE]\nLPOSE_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"pose\" in col and int(col[-2:]) in LPOSE]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:00.867191Z","iopub.execute_input":"2025-05-25T16:16:00.867422Z","iopub.status.idle":"2025-05-25T16:16:00.875683Z","shell.execute_reply.started":"2025-05-25T16:16:00.867404Z","shell.execute_reply":"2025-05-25T16:16:00.87459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Số cột (features):\", len(FEATURE_COLUMNS) + 1)  # +1 vì có 'phrase'\nprint(\"Tên các cột:\", FEATURE_COLUMNS + [\"phrase\"])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:00.877852Z","iopub.execute_input":"2025-05-25T16:16:00.878095Z","iopub.status.idle":"2025-05-25T16:16:00.901131Z","shell.execute_reply.started":"2025-05-25T16:16:00.87807Z","shell.execute_reply":"2025-05-25T16:16:00.90006Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"(21+21+10)*3+1=157","metadata":{}},{"cell_type":"code","source":"# Hàm để parse một record\ndef _parse_function(example_proto):\n    # Tạo dict schema để giải mã\n    feature_description = {\n        col: tf.io.VarLenFeature(tf.float32) for col in FEATURE_COLUMNS\n    }\n    feature_description[\"phrase\"] = tf.io.FixedLenFeature([], tf.string)\n\n    return tf.io.parse_single_example(example_proto, feature_description)\n\n# Đọc file .tfrecord (thay bằng tên file thực tế của bạn)\ntfrecord_path = '/kaggle/input/pre-data-fsp/new_data/1019715464.tfrecord'\nraw_dataset = tf.data.TFRecordDataset(tfrecord_path)\nparsed_dataset = raw_dataset.map(_parse_function)\n# In dữ liệu\nfor i, parsed_record in enumerate(parsed_dataset.take(1)): \n    print(f\"\\n🧾 Record {i+1}\")\n    print(len(parsed_record.keys()))\n    for key in FEATURE_COLUMNS:\n        values = tf.sparse.to_dense(parsed_record[key])\n        print(f\"{key}: {values.numpy().shape}\")\n    print(f\"phrase: {parsed_record['phrase'].numpy().decode('utf-8')}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:00.902229Z","iopub.execute_input":"2025-05-25T16:16:00.9026Z","iopub.status.idle":"2025-05-25T16:16:01.763012Z","shell.execute_reply.started":"2025-05-25T16:16:00.902566Z","shell.execute_reply":"2025-05-25T16:16:01.761999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"FRAME_LEN = 128","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:01.766121Z","iopub.execute_input":"2025-05-25T16:16:01.766416Z","iopub.status.idle":"2025-05-25T16:16:01.770931Z","shell.execute_reply.started":"2025-05-25T16:16:01.766393Z","shell.execute_reply":"2025-05-25T16:16:01.769698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open (\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\", \"r\") as f:\n    char_to_num = json.load(f)\n\n# Add pad_token, start pointer and end pointer to the dict\npad_token = 'P'\nstart_token = '<'\nend_token = '>'\npad_token_idx = 59\nstart_token_idx = 60\nend_token_idx = 61\n\nchar_to_num[pad_token] = pad_token_idx\nchar_to_num[start_token] = start_token_idx\nchar_to_num[end_token] = end_token_idx\nnum_to_char = {j:i for i,j in char_to_num.items()}\nprint(num_to_char)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:01.771905Z","iopub.execute_input":"2025-05-25T16:16:01.772261Z","iopub.status.idle":"2025-05-25T16:16:01.794862Z","shell.execute_reply.started":"2025-05-25T16:16:01.772239Z","shell.execute_reply":"2025-05-25T16:16:01.793756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/code/irohith/aslfr-transformer/notebook\n\n# Function to resize and add padding.\ndef resize_pad(x):\n    if tf.shape(x)[0] < FRAME_LEN:\n        x = tf.pad(x, ([[0, FRAME_LEN-tf.shape(x)[0]], [0, 0], [0, 0]]))\n    else:\n        x = tf.image.resize(x, (FRAME_LEN, tf.shape(x)[1]))\n    return x\n\n# Detect the dominant hand from the number of NaN values.\n# Dominant hand will have less NaN values since it is in frame moving.\ndef pre_process(x):\n    print(x.shape)\n    rhand = tf.gather(x, RHAND_IDX, axis=1)\n    lhand = tf.gather(x, LHAND_IDX, axis=1)\n    rpose = tf.gather(x, RPOSE_IDX, axis=1)\n    lpose = tf.gather(x, LPOSE_IDX, axis=1)\n    \n    rnan_idx = tf.reduce_any(tf.math.is_nan(rhand), axis=1)\n    lnan_idx = tf.reduce_any(tf.math.is_nan(lhand), axis=1)\n    \n    rnans = tf.math.count_nonzero(rnan_idx)\n    lnans = tf.math.count_nonzero(lnan_idx)\n    \n    # For dominant hand\n    if rnans > lnans:\n        hand = lhand\n        pose = lpose\n        \n        hand_x = hand[:, 0*(len(LHAND_IDX)//3) : 1*(len(LHAND_IDX)//3)]\n        hand_y = hand[:, 1*(len(LHAND_IDX)//3) : 2*(len(LHAND_IDX)//3)]\n        hand_z = hand[:, 2*(len(LHAND_IDX)//3) : 3*(len(LHAND_IDX)//3)]\n        hand = tf.concat([1-hand_x, hand_y, hand_z], axis=1)\n        \n        pose_x = pose[:, 0*(len(LPOSE_IDX)//3) : 1*(len(LPOSE_IDX)//3)]\n        pose_y = pose[:, 1*(len(LPOSE_IDX)//3) : 2*(len(LPOSE_IDX)//3)]\n        pose_z = pose[:, 2*(len(LPOSE_IDX)//3) : 3*(len(LPOSE_IDX)//3)]\n        pose = tf.concat([1-pose_x, pose_y, pose_z], axis=1)\n    else:\n        hand = rhand\n        pose = rpose\n    \n    hand_x = hand[:, 0*(len(LHAND_IDX)//3) : 1*(len(LHAND_IDX)//3)]\n    hand_y = hand[:, 1*(len(LHAND_IDX)//3) : 2*(len(LHAND_IDX)//3)]\n    hand_z = hand[:, 2*(len(LHAND_IDX)//3) : 3*(len(LHAND_IDX)//3)]\n    hand = tf.concat([hand_x[..., tf.newaxis], hand_y[..., tf.newaxis], hand_z[..., tf.newaxis]], axis=-1)\n    \n    mean = tf.math.reduce_mean(hand, axis=1)[:, tf.newaxis, :]\n    std = tf.math.reduce_std(hand, axis=1)[:, tf.newaxis, :]\n    hand = (hand - mean) / std\n\n    pose_x = pose[:, 0*(len(LPOSE_IDX)//3) : 1*(len(LPOSE_IDX)//3)]\n    pose_y = pose[:, 1*(len(LPOSE_IDX)//3) : 2*(len(LPOSE_IDX)//3)]\n    pose_z = pose[:, 2*(len(LPOSE_IDX)//3) : 3*(len(LPOSE_IDX)//3)]\n    pose = tf.concat([pose_x[..., tf.newaxis], pose_y[..., tf.newaxis], pose_z[..., tf.newaxis]], axis=-1)\n    \n    x = tf.concat([hand, pose], axis=1)\n    x = resize_pad(x)\n    \n    x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n    x = tf.reshape(x, (FRAME_LEN, len(LHAND_IDX) + len(LPOSE_IDX)))\n    print(x.shape)\n    return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:01.795689Z","iopub.execute_input":"2025-05-25T16:16:01.796024Z","iopub.status.idle":"2025-05-25T16:16:01.811811Z","shell.execute_reply.started":"2025-05-25T16:16:01.796Z","shell.execute_reply":"2025-05-25T16:16:01.810767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" \ndef decode_fn(record_bytes):\n    # step 1: create schema\n    schema = {COL: tf.io.VarLenFeature(dtype=tf.float32) for COL in FEATURE_COLUMNS}\n    schema[\"phrase\"] = tf.io.FixedLenFeature([], dtype=tf.string)\n\n    # step 2: Parse record\n    features = tf.io.parse_single_example(record_bytes, schema)\n    print(features[\"x_left_hand_0\"])\n    # step 3: get sequences\n    phrase = features[\"phrase\"]\n\n\n    # step 4:  SparseTensor -> Dense\n    landmarks = [tf.sparse.to_dense(features[COL]) for COL in FEATURE_COLUMNS]\n    print(\"_____________________________________________________________________\")\n    print(landmarks[0])\n    # step 5: \n    landmarks = tf.transpose(landmarks)\n    return landmarks, phrase\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:01.81301Z","iopub.execute_input":"2025-05-25T16:16:01.813292Z","iopub.status.idle":"2025-05-25T16:16:01.837053Z","shell.execute_reply.started":"2025-05-25T16:16:01.813273Z","shell.execute_reply":"2025-05-25T16:16:01.836053Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"table = tf.lookup.StaticHashTable(\n    initializer=tf.lookup.KeyValueTensorInitializer(\n        keys=list(char_to_num.keys()),\n        values=list(char_to_num.values()),\n    ),\n    default_value=tf.constant(-1),\n    name=\"class_weight\"\n)\n\ndef convert_fn(landmarks, phrase):\n    # Add start and end pointers to phrase.\n    phrase = start_token + phrase + end_token\n    phrase = tf.strings.bytes_split(phrase)\n    phrase = table.lookup(phrase)\n    # Vectorize and add padding.\n    phrase = tf.pad(phrase, paddings=[[0, 64 - tf.shape(phrase)[0]]], mode = 'CONSTANT',\n                    constant_values = pad_token_idx)\n    # Apply pre_process function to the landmarks.\n    return pre_process(landmarks), phrase","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:01.84132Z","iopub.execute_input":"2025-05-25T16:16:01.841664Z","iopub.status.idle":"2025-05-25T16:16:01.860171Z","shell.execute_reply.started":"2025-05-25T16:16:01.84163Z","shell.execute_reply":"2025-05-25T16:16:01.859067Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**example**","metadata":{}},{"cell_type":"code","source":"a = tf.strings.bytes_split(\"test bytes split\")\nprint(a)\na = table.lookup(a)\nprint(a)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:01.861183Z","iopub.execute_input":"2025-05-25T16:16:01.861528Z","iopub.status.idle":"2025-05-25T16:16:01.891992Z","shell.execute_reply.started":"2025-05-25T16:16:01.861492Z","shell.execute_reply":"2025-05-25T16:16:01.891049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"indices = [[0],[2]]\nvalues = [1.0, 3.0]\ndense_shape = [3]\n\n# Tạo SparseTensor\nsparse_tensor = tf.sparse.SparseTensor(indices, values, dense_shape)\n\n# Chuyển SparseTensor thành DenseTensor\ndense_tensor = tf.sparse.to_dense(sparse_tensor)\n\n# In kết quả\nprint(dense_tensor)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:01.892924Z","iopub.execute_input":"2025-05-25T16:16:01.89316Z","iopub.status.idle":"2025-05-25T16:16:01.901192Z","shell.execute_reply.started":"2025-05-25T16:16:01.893142Z","shell.execute_reply":"2025-05-25T16:16:01.900224Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Proceed**","metadata":{}},{"cell_type":"code","source":"batch_size = 64\ntrain_len = int(0.8 * len(tf_records))\n\ntrain_ds = tf.data.TFRecordDataset(tf_records[:train_len]).map(decode_fn).map(convert_fn).batch(batch_size).prefetch(buffer_size=tf.data.AUTOTUNE).cache()\nvalid_ds = tf.data.TFRecordDataset(tf_records[train_len:]).map(decode_fn).map(convert_fn).batch(batch_size).prefetch(buffer_size=tf.data.AUTOTUNE).cache()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:01.902733Z","iopub.execute_input":"2025-05-25T16:16:01.903042Z","iopub.status.idle":"2025-05-25T16:16:03.793978Z","shell.execute_reply.started":"2025-05-25T16:16:01.903021Z","shell.execute_reply":"2025-05-25T16:16:03.793098Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"references:https://www.tensorflow.org/api_docs/python/tf/data/TFRecordDataset","metadata":{}},{"cell_type":"code","source":"print(train_ds)\nprint(valid_ds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:03.796086Z","iopub.execute_input":"2025-05-25T16:16:03.796376Z","iopub.status.idle":"2025-05-25T16:16:03.801681Z","shell.execute_reply.started":"2025-05-25T16:16:03.796354Z","shell.execute_reply":"2025-05-25T16:16:03.800533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for batch in train_ds.take(1):\n    print(batch[0].shape)\n    print(batch[1].shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:03.802757Z","iopub.execute_input":"2025-05-25T16:16:03.803117Z","iopub.status.idle":"2025-05-25T16:16:04.319974Z","shell.execute_reply.started":"2025-05-25T16:16:03.803094Z","shell.execute_reply":"2025-05-25T16:16:04.319038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfor i, batch in enumerate(train_ds.take(5)):  # Lấy 5 batch đầu tiên làm ví dụ\n    inputs, labels = batch\n    # In số batch, số phần tử trong batch và kích thước của phần tử\n    print(f\"Batch {i + 1}:\")\n    print(inputs.shape)\n    print(labels.shape)\n    print(f\"  - Số phần tử trong batch: {inputs.shape[0]}\")\n    print(f\"  - Kích thước của mỗi phần tử trong batch: {inputs.shape[1:]}\")\n    print(f\"  - Kích thước labels: {labels.shape[1:]}\")\n    print(\"-\" * 50)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:04.322752Z","iopub.execute_input":"2025-05-25T16:16:04.323013Z","iopub.status.idle":"2025-05-25T16:16:05.152705Z","shell.execute_reply.started":"2025-05-25T16:16:04.322993Z","shell.execute_reply":"2025-05-25T16:16:05.151744Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# MODEL","metadata":{}},{"cell_type":"code","source":"class TokenEmbedding(layers.Layer):\n    def __init__(self, num_vocab=1000, maxlen=100, num_hid=64):\n        super().__init__()\n        self.emb = tf.keras.layers.Embedding(num_vocab, num_hid)\n        self.pos_emb = layers.Embedding(input_dim=maxlen, output_dim=num_hid)\n\n    def call(self, x):\n        maxlen = tf.shape(x)[-1]\n        x = self.emb(x)\n        positions = tf.range(start=0, limit=maxlen, delta=1)\n        positions = self.pos_emb(positions)\n        return x + positions\n\n\nclass LandmarkEmbedding(layers.Layer):\n    def __init__(self, num_hid=64, maxlen=100):\n        super().__init__()\n        self.conv1 = tf.keras.layers.Conv1D(\n            num_hid, 11, strides=2, padding=\"same\", activation=\"relu\"\n        )\n        self.conv2 = tf.keras.layers.Conv1D(\n            num_hid, 11, strides=2, padding=\"same\", activation=\"relu\"\n        )\n        self.conv3 = tf.keras.layers.Conv1D(\n            num_hid, 11, strides=2, padding=\"same\", activation=\"relu\"\n        )\n        self.pos_emb = layers.Embedding(input_dim=maxlen, output_dim=num_hid)\n\n    def call(self, x):\n        x = self.conv1(x)\n        x = self.conv2(x)\n        return self.conv3(x)\n    def build(self, input_shape):\n        super().build(input_shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:05.155568Z","iopub.execute_input":"2025-05-25T16:16:05.155828Z","iopub.status.idle":"2025-05-25T16:16:05.165774Z","shell.execute_reply.started":"2025-05-25T16:16:05.155809Z","shell.execute_reply":"2025-05-25T16:16:05.164628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Tạo một input mẫu (batch size = 1, sequence length = 5)\nsample_input = tf.constant([[5, 2, 8, 0, 1],[1, 2, 3, 4, 5]])\n\n# Khởi tạo lớp embedding\ntoken_emb_layer = TokenEmbedding(num_vocab=1000, maxlen=100, num_hid=16)  # 16 chiều dễ quan sát\n\n# Chạy forward pass\noutput = token_emb_layer(sample_input)\n\n# In kết quả\nprint(\"Shape của output:\", output.shape)\nprint(\"Output (đoạn đầu):\\n\", output[0, :2])  # In 2 token đầu để xem rõ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:05.166697Z","iopub.execute_input":"2025-05-25T16:16:05.166951Z","iopub.status.idle":"2025-05-25T16:16:05.273802Z","shell.execute_reply.started":"2025-05-25T16:16:05.166933Z","shell.execute_reply":"2025-05-25T16:16:05.272789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TransformerEncoder(layers.Layer):\n    def __init__(self, embed_dim, num_heads, feed_forward_dim, rate=0.1):\n        super().__init__()\n        self.att = layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        self.ffn = keras.Sequential(\n            [\n                layers.Dense(feed_forward_dim, activation=\"relu\"),\n                layers.Dense(embed_dim),\n            ]\n        )\n        self.layernorm1 = layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm2 = layers.LayerNormalization(epsilon=1e-6)\n        self.dropout1 = layers.Dropout(rate)\n        self.dropout2 = layers.Dropout(rate)\n\n    def call(self, inputs, training=False): # Thêm giá trị mặc định training=False\n        attn_output = self.att(inputs, inputs)\n        attn_output = self.dropout1(attn_output, training=training)\n        out1 = self.layernorm1(inputs + attn_output)\n        ffn_output = self.ffn(out1)\n        ffn_output = self.dropout2(ffn_output, training=training)\n        return self.layernorm2(out1 + ffn_output)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:05.274803Z","iopub.execute_input":"2025-05-25T16:16:05.275865Z","iopub.status.idle":"2025-05-25T16:16:05.28304Z","shell.execute_reply.started":"2025-05-25T16:16:05.275837Z","shell.execute_reply":"2025-05-25T16:16:05.282118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TransformerDecoder(layers.Layer):\n    def __init__(self, embed_dim, num_heads, feed_forward_dim, dropout_rate=0.1):\n        super().__init__()\n        self.layernorm1 = layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm2 = layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm3 = layers.LayerNormalization(epsilon=1e-6)\n        self.self_att = layers.MultiHeadAttention(\n            num_heads=num_heads, key_dim=embed_dim\n        )\n        self.enc_att = layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        self.self_dropout = layers.Dropout(0.5)\n        self.enc_dropout = layers.Dropout(0.1)\n        self.ffn_dropout = layers.Dropout(0.1)\n        self.ffn = keras.Sequential(\n            [\n                layers.Dense(feed_forward_dim, activation=\"relu\"),\n                layers.Dense(embed_dim),\n            ]\n        )\n\n    def causal_attention_mask(self, batch_size, n_dest, n_src, dtype):\n        \"\"\"Masks the upper half of the dot product matrix in self attention.\n\n        This prevents flow of information from future tokens to current token.\n        1's in the lower triangle, counting from the lower right corner.\n        \"\"\"\n        i = tf.range(n_dest)[:, None]\n        j = tf.range(n_src)\n        m = i >= j - n_src + n_dest\n        mask = tf.cast(m, dtype)\n        mask = tf.reshape(mask, [1, n_dest, n_src])\n        mult = tf.concat(\n            [batch_size[..., tf.newaxis], tf.constant([1, 1], dtype=tf.int32)], 0\n        )\n        return tf.tile(mask, mult)\n\n    def call(self, enc_out, target, training=False): # Thêm giá trị mặc định training=False\n        input_shape = tf.shape(target)\n        batch_size = input_shape[0]\n        seq_len = input_shape[1]\n        causal_mask = self.causal_attention_mask(batch_size, seq_len, seq_len, tf.bool)\n        target_att = self.self_att(target, target, attention_mask=causal_mask)\n        target_norm = self.layernorm1(target + self.self_dropout(target_att, training = training))\n        enc_out = self.enc_att(target_norm, enc_out)\n        enc_out_norm = self.layernorm2(self.enc_dropout(enc_out, training = training) + target_norm)\n        ffn_out = self.ffn(enc_out_norm)\n        ffn_out_norm = self.layernorm3(enc_out_norm + self.ffn_dropout(ffn_out, training = training))\n        return ffn_out_norm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:05.283928Z","iopub.execute_input":"2025-05-25T16:16:05.284218Z","iopub.status.idle":"2025-05-25T16:16:05.308203Z","shell.execute_reply.started":"2025-05-25T16:16:05.284198Z","shell.execute_reply":"2025-05-25T16:16:05.307141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TransformerEncoderStack(keras.layers.Layer):\n    def __init__(self, num_layers, num_hid, num_head, num_feed_forward):\n        super().__init__()\n        self.enc_layers = [\n            TransformerEncoder(num_hid, num_head, num_feed_forward)\n            for _ in range(num_layers)\n        ]\n\n    def call(self, x, training=False): # Đã có training=False, không cần sửa\n        for layer in self.enc_layers:\n            x = layer(x, training=training)\n        return x\n\n\nclass Transformer(keras.Model):\n    def __init__(\n        self,\n        num_hid=64,\n        num_head=2,\n        num_feed_forward=128,\n        source_maxlen=100,\n        target_maxlen=100,\n        num_layers_enc=4,\n        num_layers_dec=1,\n        num_classes=60,\n    ):\n        super().__init__()\n        self.loss_metric = keras.metrics.Mean(name=\"loss\")\n        self.acc_metric = keras.metrics.Mean(name=\"edit_dist\")\n        self.num_layers_enc = num_layers_enc\n        self.num_layers_dec = num_layers_dec\n        self.target_maxlen = target_maxlen\n        self.num_classes = num_classes\n\n        self.enc_input = LandmarkEmbedding(num_hid=num_hid, maxlen=source_maxlen)\n        self.dec_input = TokenEmbedding(\n            num_vocab=num_classes, maxlen=target_maxlen, num_hid=num_hid\n        )\n\n        self.encoder = TransformerEncoderStack(\n            num_layers=num_layers_enc,\n            num_hid=num_hid,\n            num_head=num_head,\n            num_feed_forward=num_feed_forward\n        )\n\n        for i in range(num_layers_dec):\n            setattr(\n                self,\n                f\"dec_layer_{i}\",\n                TransformerDecoder(num_hid, num_head, num_feed_forward),\n            )\n\n        self.classifier = layers.Dense(num_classes)\n\n    def decode(self, enc_out, target, training):\n        y = self.dec_input(target)\n        for i in range(self.num_layers_dec):\n            y = getattr(self, f\"dec_layer_{i}\")(enc_out, y, training=training)\n        return y\n\n    def call(self, inputs, training=False): # Thêm giá trị mặc định training=False\n        source = inputs[0]\n        target = inputs[1]\n        x = self.enc_input(source)  # gọi embedding trước\n        x = self.encoder(x, training=training)\n        y = self.decode(x, target, training=training)\n        return self.classifier(y)\n\n    @property\n    def metrics(self):\n        return [self.loss_metric]\n\n    def train_step(self, batch):\n        source = batch[0]\n        target = batch[1]\n\n        dec_input = target[:, :-1]\n        dec_target = target[:, 1:]\n\n        with tf.GradientTape() as tape:\n            preds = self([source, dec_input], training=True)\n            one_hot = tf.one_hot(dec_target, depth=self.num_classes)\n            mask = tf.math.logical_not(tf.math.equal(dec_target, pad_token_idx))\n            loss = self.compiled_loss(one_hot, preds, sample_weight=mask)\n\n        gradients = tape.gradient(loss, self.trainable_variables)\n        self.optimizer.apply_gradients(zip(gradients, self.trainable_variables))\n\n        edit_dist = tf.edit_distance(\n            tf.sparse.from_dense(target),\n            tf.sparse.from_dense(tf.cast(tf.argmax(preds, axis=1), tf.int32))\n        )\n        self.acc_metric.update_state(tf.reduce_mean(edit_dist))\n        self.loss_metric.update_state(loss)\n        return {\"loss\": self.loss_metric.result(), \"edit_dist\": self.acc_metric.result()}\n\n    def test_step(self, batch):\n        source = batch[0]\n        target = batch[1]\n\n        dec_input = target[:, :-1]\n        dec_target = target[:, 1:]\n\n        preds = self([source, dec_input], training=False)\n        one_hot = tf.one_hot(dec_target, depth=self.num_classes)\n        mask = tf.math.logical_not(tf.math.equal(dec_target, pad_token_idx))\n        loss = self.compiled_loss(one_hot, preds, sample_weight=mask)\n\n        edit_dist = tf.edit_distance(\n            tf.sparse.from_dense(target),\n            tf.sparse.from_dense(tf.cast(tf.argmax(preds, axis=1), tf.int32))\n        )\n        self.acc_metric.update_state(tf.reduce_mean(edit_dist))\n        self.loss_metric.update_state(loss)\n        return {\"loss\": self.loss_metric.result(), \"edit_dist\": self.acc_metric.result()}\n\n    def generate(self, source, target_start_token_idx):\n        bs = tf.shape(source)[0]\n        x = self.enc_input(source)\n        enc = self.encoder(x, training=False) # Đã có training=False\n    \n        dec_input = tf.ones((bs, 1), dtype=tf.int32) * target_start_token_idx\n        for _ in range(self.target_maxlen - 1):\n            dec_out = self.decode(enc, dec_input, training=False) # Đã có training=False\n            logits = self.classifier(dec_out)\n            next_token = tf.argmax(logits[:, -1, :], axis=-1, output_type=tf.int32)\n            next_token = tf.expand_dims(next_token, axis=1)\n            dec_input = tf.concat([dec_input, next_token], axis=1)\n        return dec_input\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:05.309325Z","iopub.execute_input":"2025-05-25T16:16:05.309657Z","iopub.status.idle":"2025-05-25T16:16:05.33383Z","shell.execute_reply.started":"2025-05-25T16:16:05.309624Z","shell.execute_reply":"2025-05-25T16:16:05.332912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DisplayOutputs(keras.callbacks.Callback):\n    def __init__(\n        self, batch, idx_to_token, target_start_token_idx=60, target_end_token_idx=61\n    ):\n        \"\"\"Displays a batch of outputs after every 4 epoch\n\n        Args:\n            batch: A test batch\n            idx_to_token: A List containing the vocabulary tokens corresponding to their indices\n            target_start_token_idx: A start token index in the target vocabulary\n            target_end_token_idx: An end token index in the target vocabulary\n        \"\"\"\n        self.batch = batch\n        self.target_start_token_idx = target_start_token_idx\n        self.target_end_token_idx = target_end_token_idx\n        self.idx_to_char = idx_to_token\n\n    def on_epoch_end(self, epoch, logs=None):\n        if epoch % 4 != 0:\n            return\n        source = self.batch[0]\n        target = self.batch[1].numpy()\n        bs = tf.shape(source)[0]\n        preds = self.model.generate(source, self.target_start_token_idx)\n        preds = preds.numpy()\n        for i in range(bs):\n            target_text = \"\".join([self.idx_to_char[_] for _ in target[i, :]])\n            prediction = \"\"\n            for idx in preds[i, :]:\n                prediction += self.idx_to_char[idx]\n                if idx == self.target_end_token_idx:\n                    break\n            print(f\"target:     {target_text.replace('-','')}\")\n            print(f\"prediction: {prediction}\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:05.335002Z","iopub.execute_input":"2025-05-25T16:16:05.335825Z","iopub.status.idle":"2025-05-25T16:16:05.358799Z","shell.execute_reply.started":"2025-05-25T16:16:05.335775Z","shell.execute_reply":"2025-05-25T16:16:05.357819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping\nbatch = next(iter(valid_ds))\n\n# The vocabulary to convert predicted indices into characters\nidx_to_char = list(char_to_num.keys())\ndisplay_cb = DisplayOutputs(\n    batch, idx_to_char, target_start_token_idx=char_to_num['<'], target_end_token_idx=char_to_num['>']\n)  # set the arguments as per vocabulary index for '<' and '>'\nearlystop_cb = EarlyStopping(\n    monitor='val_loss',   \n    patience=3,           # dừng sau 3 epoch không cải thiện\n    restore_best_weights=True\n)\nmodel = Transformer(\n    num_hid=200,\n    num_head=4,\n    num_feed_forward=400,\n    source_maxlen = FRAME_LEN,\n    target_maxlen=64,\n    num_layers_enc=2,\n    num_layers_dec=1,\n    num_classes=62\n)\nloss_fn = tf.keras.losses.CategoricalCrossentropy(\n    from_logits=True, label_smoothing=0.1,\n)\noptimizer = keras.optimizers.Adam(0.0001)\nmodel.compile(optimizer=optimizer, loss=loss_fn)\n\nhistory = model.fit(train_ds, validation_data=valid_ds, callbacks=[display_cb,earlystop_cb], epochs=100)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:16:05.35988Z","iopub.execute_input":"2025-05-25T16:16:05.360155Z","iopub.status.idle":"2025-05-25T16:30:15.647323Z","shell.execute_reply.started":"2025-05-25T16:16:05.360134Z","shell.execute_reply":"2025-05-25T16:30:15.646551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.legend(['training loss', 'val_loss'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:30:15.648699Z","iopub.execute_input":"2025-05-25T16:30:15.649626Z","iopub.status.idle":"2025-05-25T16:30:15.842934Z","shell.execute_reply.started":"2025-05-25T16:30:15.649599Z","shell.execute_reply":"2025-05-25T16:30:15.841837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_df = pd.read_csv('/kaggle/input/asl-fingerspelling/supplemental_metadata.csv')\ntf_records = dataset_df.file_id.map(lambda x: f'/kaggle/input/test-fsl/test/{x}.tfrecord').unique()\nprint(f\"List of {len(tf_records)} TFRecord files.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:33:23.598197Z","iopub.execute_input":"2025-05-25T16:33:23.598497Z","iopub.status.idle":"2025-05-25T16:33:23.691082Z","shell.execute_reply.started":"2025-05-25T16:33:23.598453Z","shell.execute_reply":"2025-05-25T16:33:23.690168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ds = tf.data.TFRecordDataset(tf_records).map(decode_fn).map(convert_fn).batch(batch_size).prefetch(buffer_size=tf.data.AUTOTUNE).cache()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:34:01.923235Z","iopub.execute_input":"2025-05-25T16:34:01.923601Z","iopub.status.idle":"2025-05-25T16:34:02.579414Z","shell.execute_reply.started":"2025-05-25T16:34:01.923565Z","shell.execute_reply":"2025-05-25T16:34:02.57842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = model.evaluate(test_ds)\nprint(\"Loss:\", results[0])\nprint(\"Edit distance:\", results[1])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:36:54.797403Z","iopub.execute_input":"2025-05-25T16:36:54.798255Z","iopub.status.idle":"2025-05-25T16:40:21.718536Z","shell.execute_reply.started":"2025-05-25T16:36:54.798219Z","shell.execute_reply":"2025-05-25T16:40:21.717638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for (images, labels) in train_ds.take(1):\n    # nếu labels là chuỗi số (token index)\n    for i in range(3):  # in thử 3 mẫu\n        label = labels[i].numpy()\n        decoded = ''.join([num_to_char[idx] for idx in label if idx != 0])\n        print(f\"Decoded target string {i+1}: {decoded}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:45:44.637637Z","iopub.execute_input":"2025-05-25T16:45:44.638644Z","iopub.status.idle":"2025-05-25T16:45:44.759289Z","shell.execute_reply.started":"2025-05-25T16:45:44.63861Z","shell.execute_reply":"2025-05-25T16:45:44.758434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"start_token = char_to_num['<']\nend_token = char_to_num['>']\n\nnum_to_char = {v: k for k, v in char_to_num.items()}\n\ndef decode_sequence(seq):\n    chars = []\n    for idx in seq:\n        if idx == end_token:\n            break\n        chars.append(num_to_char[idx])\n    return ''.join(chars)\n\nfor batch in test_ds.take(1):\n    source_input, target_output = batch\n    pred = model.generate(source_input, target_start_token_idx=start_token)\n\n    pred_strs = [decode_sequence(seq.numpy()[1:]) for seq in pred]\n    true_strs = [decode_sequence(seq.numpy()[1:]) for seq in target_output]\n\n    for p, t in zip(pred_strs, true_strs):\n        print(f\"▶️ Pred: {p} \\n✅ True: {t}\\n\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T16:48:11.684608Z","iopub.execute_input":"2025-05-25T16:48:11.685474Z","iopub.status.idle":"2025-05-25T16:48:26.333407Z","shell.execute_reply.started":"2025-05-25T16:48:11.685425Z","shell.execute_reply":"2025-05-25T16:48:26.332514Z"}},"outputs":[],"execution_count":null}]}