{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":52950,"databundleVersionId":5973250,"sourceType":"competition"},{"sourceId":11694503,"sourceType":"datasetVersion","datasetId":7340011},{"sourceId":240611911,"sourceType":"kernelVersion"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:06.250469Z","iopub.execute_input":"2025-05-25T17:12:06.250737Z","iopub.status.idle":"2025-05-25T17:12:06.999143Z","shell.execute_reply.started":"2025-05-25T17:12:06.250717Z","shell.execute_reply":"2025-05-25T17:12:06.99833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nimport pyarrow.parquet as pq\nimport tensorflow as tf\nimport json\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport random\n\nfrom skimage.transform import resize\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tqdm.notebook import tqdm\nfrom matplotlib import animation, rc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:07.000454Z","iopub.execute_input":"2025-05-25T17:12:07.000826Z","iopub.status.idle":"2025-05-25T17:12:20.390744Z","shell.execute_reply.started":"2025-05-25T17:12:07.000807Z","shell.execute_reply":"2025-05-25T17:12:20.390204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:20.391398Z","iopub.execute_input":"2025-05-25T17:12:20.391874Z","iopub.status.idle":"2025-05-25T17:12:20.395716Z","shell.execute_reply.started":"2025-05-25T17:12:20.39185Z","shell.execute_reply":"2025-05-25T17:12:20.395007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_df = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\ntf_records = dataset_df.file_id.map(lambda x: f'/kaggle/input/pre-data-fsp/new_data/{x}.tfrecord').unique()\nprint(f\"List of {len(tf_records)} TFRecord files.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:20.397476Z","iopub.execute_input":"2025-05-25T17:12:20.398196Z","iopub.status.idle":"2025-05-25T17:12:20.593676Z","shell.execute_reply.started":"2025-05-25T17:12:20.398172Z","shell.execute_reply":"2025-05-25T17:12:20.592835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LPOSE = [13, 15, 17, 19, 21]\nRPOSE = [14, 16, 18, 20, 22]\nPOSE = LPOSE + RPOSE\nX = [f'x_right_hand_{i}' for i in range(21)] + [f'x_left_hand_{i}' for i in range(21)] + [f'x_pose_{i}' for i in POSE]\nY = [f'y_right_hand_{i}' for i in range(21)] + [f'y_left_hand_{i}' for i in range(21)] + [f'y_pose_{i}' for i in POSE]\nZ = [f'z_right_hand_{i}' for i in range(21)] + [f'z_left_hand_{i}' for i in range(21)] + [f'z_pose_{i}' for i in POSE]\nFEATURE_COLUMNS = X + Y + Z\nRHAND_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"right\" in col]\nLHAND_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"left\" in col]\nRPOSE_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"pose\" in col and int(col[-2:]) in RPOSE]\nLPOSE_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"pose\" in col and int(col[-2:]) in LPOSE]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:20.59443Z","iopub.execute_input":"2025-05-25T17:12:20.594739Z","iopub.status.idle":"2025-05-25T17:12:20.601394Z","shell.execute_reply.started":"2025-05-25T17:12:20.594721Z","shell.execute_reply":"2025-05-25T17:12:20.600751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Số cột (features):\", len(FEATURE_COLUMNS) + 1)  # +1 vì có 'phrase'\nprint(\"Tên các cột:\", FEATURE_COLUMNS + [\"phrase\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:20.602129Z","iopub.execute_input":"2025-05-25T17:12:20.60284Z","iopub.status.idle":"2025-05-25T17:12:20.618046Z","shell.execute_reply.started":"2025-05-25T17:12:20.602816Z","shell.execute_reply":"2025-05-25T17:12:20.61736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hàm để parse một record\ndef _parse_function(example_proto):\n    # Tạo dict schema để giải mã\n    feature_description = {\n        col: tf.io.VarLenFeature(tf.float32) for col in FEATURE_COLUMNS\n    }\n    feature_description[\"phrase\"] = tf.io.FixedLenFeature([], tf.string)\n\n    return tf.io.parse_single_example(example_proto, feature_description)\n\n# Đọc file .tfrecord (thay bằng tên file thực tế của bạn)\ntfrecord_path = '/kaggle/input/pre-data-fsp/new_data/1019715464.tfrecord'\nraw_dataset = tf.data.TFRecordDataset(tfrecord_path)\nparsed_dataset = raw_dataset.map(_parse_function)\n# In dữ liệu\nfor i, parsed_record in enumerate(parsed_dataset.take(1)): \n    print(f\"\\n🧾 Record {i+1}\")\n    print(len(parsed_record.keys()))\n    for key in FEATURE_COLUMNS:\n        values = tf.sparse.to_dense(parsed_record[key])\n        print(f\"{key}: {values.numpy().shape}\")\n    print(f\"phrase: {parsed_record['phrase'].numpy().decode('utf-8')}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:20.618758Z","iopub.execute_input":"2025-05-25T17:12:20.619159Z","iopub.status.idle":"2025-05-25T17:12:22.545605Z","shell.execute_reply.started":"2025-05-25T17:12:20.619124Z","shell.execute_reply":"2025-05-25T17:12:22.544793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"FRAME_LEN = 128","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:22.546588Z","iopub.execute_input":"2025-05-25T17:12:22.54717Z","iopub.status.idle":"2025-05-25T17:12:22.550319Z","shell.execute_reply.started":"2025-05-25T17:12:22.547146Z","shell.execute_reply":"2025-05-25T17:12:22.5495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nimport tensorflow as tf # Ensure tf is imported for StaticHashTable later\n\n# Load original character to number mapping\nwith open(\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\", \"r\") as f:\n    char_to_num_orig = json.load(f)\nVOCAB_SIZE = len(char_to_num_orig)\nBLANK_LABEL_IDX = VOCAB_SIZE  \nNUM_CTC_CLASSES = VOCAB_SIZE + 1 # e.g., 60\n\n# Create num_to_char based on the original mapping\nnum_to_char = {j:i for i,j in char_to_num_orig.items()}\n\n# For padding target labels in convert_fn, we can use a value like 0 (space)\n# as long as label_length is accurate.\n# Or, define a specific PAD_VALUE for labels if space is critical and distinct from padding.\n# For this example, let's assume label_length correctly handles it, and padding with 0 is fine.\n# The char_to_num_orig already contains ' ' mapped to 0.\n\n# char_to_num will be the original mapping\nchar_to_num = char_to_num_orig\n\nprint(f\"Original Vocabulary Size (VOCAB_SIZE): {VOCAB_SIZE}\")\nprint(f\"CTC Blank Label Index (BLANK_LABEL_IDX): {BLANK_LABEL_IDX}\")\nprint(f\"Total CTC Prediction Classes (NUM_CTC_CLASSES): {NUM_CTC_CLASSES}\")\nprint(\"num_to_char mapping (first 5 entries):\")\nfor i in range(min(5, len(num_to_char))):\n    print(f\"  {i}: {num_to_char.get(i, 'N/A')}\")\n# Max phrase length for padding labels\nTARGET_MAXLEN = 64\nPAD_TOKEN_LABEL_VALUE = 0 # Using space (index 0) for padding labels, ensure label_length is accurate","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:22.55109Z","iopub.execute_input":"2025-05-25T17:12:22.551313Z","iopub.status.idle":"2025-05-25T17:12:22.581469Z","shell.execute_reply.started":"2025-05-25T17:12:22.551289Z","shell.execute_reply":"2025-05-25T17:12:22.580918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def resize_pad(x):\n    if tf.shape(x)[0] < FRAME_LEN:\n        x = tf.pad(x, ([[0, FRAME_LEN-tf.shape(x)[0]], [0, 0], [0, 0]]))\n    else:\n        x = tf.image.resize(x, (FRAME_LEN, tf.shape(x)[1]))\n    return x\n\n# Detect the dominant hand from the number of NaN values.\n# Dominant hand will have less NaN values since it is in frame moving.\ndef pre_process(x):\n    print(x.shape)\n    rhand = tf.gather(x, RHAND_IDX, axis=1)\n    lhand = tf.gather(x, LHAND_IDX, axis=1)\n    rpose = tf.gather(x, RPOSE_IDX, axis=1)\n    lpose = tf.gather(x, LPOSE_IDX, axis=1)\n    \n    rnan_idx = tf.reduce_any(tf.math.is_nan(rhand), axis=1)\n    lnan_idx = tf.reduce_any(tf.math.is_nan(lhand), axis=1)\n    \n    rnans = tf.math.count_nonzero(rnan_idx)\n    lnans = tf.math.count_nonzero(lnan_idx)\n    \n    # For dominant hand\n    if rnans > lnans:\n        hand = lhand\n        pose = lpose\n        \n        hand_x = hand[:, 0*(len(LHAND_IDX)//3) : 1*(len(LHAND_IDX)//3)]\n        hand_y = hand[:, 1*(len(LHAND_IDX)//3) : 2*(len(LHAND_IDX)//3)]\n        hand_z = hand[:, 2*(len(LHAND_IDX)//3) : 3*(len(LHAND_IDX)//3)]\n        hand = tf.concat([1-hand_x, hand_y, hand_z], axis=1)\n        \n        pose_x = pose[:, 0*(len(LPOSE_IDX)//3) : 1*(len(LPOSE_IDX)//3)]\n        pose_y = pose[:, 1*(len(LPOSE_IDX)//3) : 2*(len(LPOSE_IDX)//3)]\n        pose_z = pose[:, 2*(len(LPOSE_IDX)//3) : 3*(len(LPOSE_IDX)//3)]\n        pose = tf.concat([1-pose_x, pose_y, pose_z], axis=1)\n    else:\n        hand = rhand\n        pose = rpose\n    \n    hand_x = hand[:, 0*(len(LHAND_IDX)//3) : 1*(len(LHAND_IDX)//3)]\n    hand_y = hand[:, 1*(len(LHAND_IDX)//3) : 2*(len(LHAND_IDX)//3)]\n    hand_z = hand[:, 2*(len(LHAND_IDX)//3) : 3*(len(LHAND_IDX)//3)]\n    hand = tf.concat([hand_x[..., tf.newaxis], hand_y[..., tf.newaxis], hand_z[..., tf.newaxis]], axis=-1)\n    \n    mean = tf.math.reduce_mean(hand, axis=1)[:, tf.newaxis, :]\n    std = tf.math.reduce_std(hand, axis=1)[:, tf.newaxis, :]\n    hand = (hand - mean) / std\n\n    pose_x = pose[:, 0*(len(LPOSE_IDX)//3) : 1*(len(LPOSE_IDX)//3)]\n    pose_y = pose[:, 1*(len(LPOSE_IDX)//3) : 2*(len(LPOSE_IDX)//3)]\n    pose_z = pose[:, 2*(len(LPOSE_IDX)//3) : 3*(len(LPOSE_IDX)//3)]\n    pose = tf.concat([pose_x[..., tf.newaxis], pose_y[..., tf.newaxis], pose_z[..., tf.newaxis]], axis=-1)\n    \n    x = tf.concat([hand, pose], axis=1)\n    x = resize_pad(x)\n    \n    x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n    x = tf.reshape(x, (FRAME_LEN, len(LHAND_IDX) + len(LPOSE_IDX)))\n    print(x.shape)\n    return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:22.58365Z","iopub.execute_input":"2025-05-25T17:12:22.583845Z","iopub.status.idle":"2025-05-25T17:12:22.602938Z","shell.execute_reply.started":"2025-05-25T17:12:22.583831Z","shell.execute_reply":"2025-05-25T17:12:22.602318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def decode_fn(record_bytes):\n    # step 1: create schema\n    schema = {COL: tf.io.VarLenFeature(dtype=tf.float32) for COL in FEATURE_COLUMNS}\n    schema[\"phrase\"] = tf.io.FixedLenFeature([], dtype=tf.string)\n\n    # step 2: Parse record\n    features = tf.io.parse_single_example(record_bytes, schema)\n    print(features[\"x_left_hand_0\"])\n    # step 3: get sequences\n    phrase = features[\"phrase\"]\n\n\n    # step 4:  SparseTensor -> Dense\n    landmarks = [tf.sparse.to_dense(features[COL]) for COL in FEATURE_COLUMNS]\n    print(\"_____________________________________________________________________\")\n    print(landmarks[0])\n    # step 5: \n    landmarks = tf.transpose(landmarks)\n    return landmarks, phrase","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:22.60356Z","iopub.execute_input":"2025-05-25T17:12:22.603739Z","iopub.status.idle":"2025-05-25T17:12:22.622073Z","shell.execute_reply.started":"2025-05-25T17:12:22.603725Z","shell.execute_reply":"2025-05-25T17:12:22.62137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"table = tf.lookup.StaticHashTable(\n    initializer=tf.lookup.KeyValueTensorInitializer(\n        keys=list(char_to_num.keys()), # Use the original char_to_num\n        values=list(char_to_num.values()),\n    ),\n    default_value=tf.constant(-1), # Or a more robust error value if a char is not found\n    name=\"character_lookup_table\"\n)\n\ndef convert_fn(landmarks, phrase_str_tensor):\n    # phrase_str_tensor is a scalar string tensor\n    phrase_chars = tf.strings.bytes_split(phrase_str_tensor)\n    \n    # Calculate actual label length BEFORE padding\n    label_length = tf.shape(phrase_chars)[0]\n    \n    phrase_indices = table.lookup(phrase_chars)\n    \n    # Pad phrase_indices to TARGET_MAXLEN\n    # Important: CTC loss needs labels to be in [0, NUM_CTC_CLASSES - 2]\n    # Pad with a value that can be handled by label_length, e.g., 0 (space)\n    phrase_indices_padded = tf.pad(phrase_indices,\n                                   paddings=[[0, TARGET_MAXLEN - tf.shape(phrase_indices)[0]]],\n                                   mode='CONSTANT',\n                                   constant_values=PAD_TOKEN_LABEL_VALUE) # e.g., space character index 0\n    \n    # Apply pre_process function to the landmarks.\n    processed_landmarks = pre_process(landmarks) # pre_process must be defined and accessible\n    \n    return processed_landmarks, phrase_indices_padded, tf.cast(label_length, tf.int32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:22.622751Z","iopub.execute_input":"2025-05-25T17:12:22.623004Z","iopub.status.idle":"2025-05-25T17:12:22.647531Z","shell.execute_reply.started":"2025-05-25T17:12:22.622977Z","shell.execute_reply":"2025-05-25T17:12:22.646922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"a = tf.strings.bytes_split(\"test bytes split\")\nprint(a)\na = table.lookup(a)\nprint(a)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:22.648189Z","iopub.execute_input":"2025-05-25T17:12:22.648348Z","iopub.status.idle":"2025-05-25T17:12:22.727762Z","shell.execute_reply.started":"2025-05-25T17:12:22.648335Z","shell.execute_reply":"2025-05-25T17:12:22.726942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"indices = [[0],[2]]\nvalues = [1.0, 3.0]\ndense_shape = [3]\n# Tạo SparseTensor\nsparse_tensor = tf.sparse.SparseTensor(indices, values, dense_shape)\n# Chuyển SparseTensor thành DenseTensor\ndense_tensor = tf.sparse.to_dense(sparse_tensor)\n# In kết quả\nprint(dense_tensor)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:22.728559Z","iopub.execute_input":"2025-05-25T17:12:22.72893Z","iopub.status.idle":"2025-05-25T17:12:22.735825Z","shell.execute_reply.started":"2025-05-25T17:12:22.728906Z","shell.execute_reply":"2025-05-25T17:12:22.735227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batch_size = 64\ntrain_len = int(0.8 * len(tf_records))\n\n# AUTOTUNE for prefetch\nAUTOTUNE = tf.data.AUTOTUNE\n\n# Note: decode_fn processes TFRecord into (landmarks, phrase_string)\n# convert_fn processes (landmarks, phrase_string) into (processed_landmarks, phrase_indices_padded, label_length)\n\ntrain_ds = tf.data.TFRecordDataset(tf_records[:train_len]) \\\n    .map(decode_fn, num_parallel_calls=AUTOTUNE) \\\n    .map(convert_fn, num_parallel_calls=AUTOTUNE) \\\n    .batch(batch_size) \\\n    .prefetch(buffer_size=AUTOTUNE) \\\n    .cache()\n\nvalid_ds = tf.data.TFRecordDataset(tf_records[train_len:]) \\\n    .map(decode_fn, num_parallel_calls=AUTOTUNE) \\\n    .map(convert_fn, num_parallel_calls=AUTOTUNE) \\\n    .batch(batch_size) \\\n    .prefetch(buffer_size=AUTOTUNE) \\\n    .cache()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:22.736508Z","iopub.execute_input":"2025-05-25T17:12:22.736723Z","iopub.status.idle":"2025-05-25T17:12:24.185944Z","shell.execute_reply.started":"2025-05-25T17:12:22.736703Z","shell.execute_reply":"2025-05-25T17:12:24.185201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_ds)\nprint(valid_ds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:24.187033Z","iopub.execute_input":"2025-05-25T17:12:24.187302Z","iopub.status.idle":"2025-05-25T17:12:24.191272Z","shell.execute_reply.started":"2025-05-25T17:12:24.187275Z","shell.execute_reply":"2025-05-25T17:12:24.190435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfor batch in train_ds.take(1):\n    print(\"Landmarks shape:\", batch[0].shape)\n    print(\"Phrase indices shape:\", batch[1].shape)\n    print(\"Label lengths shape:\", batch[2].shape)\n\n# Cell 17 (example modification)\nfor i, batch in enumerate(train_ds.take(5)):\n    inputs, label_indices, label_lengths = batch\n    print(f\"Batch {i + 1}:\")\n    print(f\"  Inputs shape: {inputs.shape}\")\n    print(f\"  Label indices shape: {label_indices.shape}\")\n    print(f\"  Label lengths shape: {label_lengths.shape}\")\n    print(f\"  Number of elements in batch: {inputs.shape[0]}\")\n    print(f\"  Processed landmark sequence length: {inputs.shape[1]}\") # Should be FRAME_LEN (e.g. 128)\n    print(f\"  Processed landmark feature dim: {inputs.shape[2]}\") # Should be 78\n    print(f\"  Padded label max length: {label_indices.shape[1]}\") # Should be TARGET_MAXLEN (e.g. 64)\n    print(f\"  Example label lengths in batch: {label_lengths[:5].numpy()}\")\n    print(\"-\" * 50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:24.191974Z","iopub.execute_input":"2025-05-25T17:12:24.192246Z","iopub.status.idle":"2025-05-25T17:12:25.236427Z","shell.execute_reply.started":"2025-05-25T17:12:24.192231Z","shell.execute_reply":"2025-05-25T17:12:25.235675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TokenEmbedding(layers.Layer):\n    def __init__(self, num_vocab=1000, maxlen=100, num_hid=64):\n        super().__init__()\n        self.emb = tf.keras.layers.Embedding(num_vocab, num_hid)\n        self.pos_emb = layers.Embedding(input_dim=maxlen, output_dim=num_hid)\n\n    def call(self, x):\n        maxlen = tf.shape(x)[-1]\n        x = self.emb(x)\n        positions = tf.range(start=0, limit=maxlen, delta=1)\n        positions = self.pos_emb(positions)\n        return x + positions\n\n\nclass LandmarkEmbedding(layers.Layer):\n    def __init__(self, num_hid=64, maxlen=FRAME_LEN):\n        super().__init__()\n        self.conv1 = tf.keras.layers.Conv1D( # Giảm chiều dài 1 lần\n            num_hid, 11, strides=2, padding=\"same\", activation=\"relu\"\n        )\n        self.conv2 = tf.keras.layers.Conv1D(\n            num_hid, 11, strides=1, padding=\"same\", activation=\"relu\"\n        )\n        self.conv3 = tf.keras.layers.Conv1D(\n            num_hid, 11, strides=1, padding=\"same\", activation=\"relu\"\n        )\n    def call(self, x):\n        x = self.conv1(x)\n        x = self.conv2(x)\n        return self.conv3(x)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:25.238151Z","iopub.execute_input":"2025-05-25T17:12:25.238417Z","iopub.status.idle":"2025-05-25T17:12:25.248467Z","shell.execute_reply.started":"2025-05-25T17:12:25.23839Z","shell.execute_reply":"2025-05-25T17:12:25.247721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Tạo một input mẫu (batch size = 1, sequence length = 5)\nsample_input = tf.constant([[5, 2, 8, 0, 1],[1, 2, 3, 4, 5]])\n\n# Khởi tạo lớp embedding\ntoken_emb_layer = TokenEmbedding(num_vocab=1000, maxlen=100, num_hid=16)  # 16 chiều dễ quan sát\n\n# Chạy forward pass\noutput = token_emb_layer(sample_input)\n\n# In kết quả\nprint(\"Shape của output:\", output.shape)\nprint(\"Output (đoạn đầu):\\n\", output[0, :2])  # In 2 token đầu để xem rõ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:25.249298Z","iopub.execute_input":"2025-05-25T17:12:25.249545Z","iopub.status.idle":"2025-05-25T17:12:27.167856Z","shell.execute_reply.started":"2025-05-25T17:12:25.249531Z","shell.execute_reply":"2025-05-25T17:12:27.167214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ENCODER_OUTPUT_SEQ_LEN = FRAME_LEN // 2\nFRAME_LEN = 128\nBLANK_LABEL_IDX = 59 # Giả sử VOCAB_SIZE = 59\nNUM_CTC_CLASSES = 60\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:27.168618Z","iopub.execute_input":"2025-05-25T17:12:27.168944Z","iopub.status.idle":"2025-05-25T17:12:27.172452Z","shell.execute_reply.started":"2025-05-25T17:12:27.168922Z","shell.execute_reply":"2025-05-25T17:12:27.171748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TransformerEncoder(layers.Layer): # Bạn cần định nghĩa lớp này từ code gốc\n    def __init__(self, embed_dim, num_heads, feed_forward_dim, rate=0.1):\n        super().__init__()\n        self.att = layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        self.ffn = keras.Sequential(\n            [\n                layers.Dense(feed_forward_dim, activation=\"relu\"),\n                layers.Dense(embed_dim),\n            ]\n        )\n        self.layernorm1 = layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm2 = layers.LayerNormalization(epsilon=1e-6)\n        self.dropout1 = layers.Dropout(rate)\n        self.dropout2 = layers.Dropout(rate)\n\n    def call(self, inputs, training=False):\n        attn_output = self.att(inputs, inputs)\n        attn_output = self.dropout1(attn_output, training=training)\n        out1 = self.layernorm1(inputs + attn_output)\n        ffn_output = self.ffn(out1)\n        ffn_output = self.dropout2(ffn_output, training=training)\n        return self.layernorm2(out1 + ffn_output)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:27.173226Z","iopub.execute_input":"2025-05-25T17:12:27.173491Z","iopub.status.idle":"2025-05-25T17:12:27.187194Z","shell.execute_reply.started":"2025-05-25T17:12:27.173469Z","shell.execute_reply":"2025-05-25T17:12:27.186607Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TransformerDecoder(layers.Layer):\n    def __init__(self, embed_dim, num_heads, feed_forward_dim, dropout_rate=0.1):\n        super().__init__()\n        self.layernorm1 = layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm2 = layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm3 = layers.LayerNormalization(epsilon=1e-6)\n        self.self_att = layers.MultiHeadAttention(\n            num_heads=num_heads, key_dim=embed_dim\n        )\n        self.enc_att = layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        self.self_dropout = layers.Dropout(0.5)\n        self.enc_dropout = layers.Dropout(0.1)\n        self.ffn_dropout = layers.Dropout(0.1)\n        self.ffn = keras.Sequential(\n            [\n                layers.Dense(feed_forward_dim, activation=\"relu\"),\n                layers.Dense(embed_dim),\n            ]\n        )\n\n    def causal_attention_mask(self, batch_size, n_dest, n_src, dtype):\n        \"\"\"Masks the upper half of the dot product matrix in self attention.\n\n        This prevents flow of information from future tokens to current token.\n        1's in the lower triangle, counting from the lower right corner.\n        \"\"\"\n        i = tf.range(n_dest)[:, None]\n        j = tf.range(n_src)\n        m = i >= j - n_src + n_dest\n        mask = tf.cast(m, dtype)\n        mask = tf.reshape(mask, [1, n_dest, n_src])\n        mult = tf.concat(\n            [batch_size[..., tf.newaxis], tf.constant([1, 1], dtype=tf.int32)], 0\n        )\n        return tf.tile(mask, mult)\n\n    def call(self, enc_out, target, training):\n        input_shape = tf.shape(target)\n        batch_size = input_shape[0]\n        seq_len = input_shape[1]\n        causal_mask = self.causal_attention_mask(batch_size, seq_len, seq_len, tf.bool)\n        target_att = self.self_att(target, target, attention_mask=causal_mask)\n        target_norm = self.layernorm1(target + self.self_dropout(target_att, training = training))\n        enc_out = self.enc_att(target_norm, enc_out)\n        enc_out_norm = self.layernorm2(self.enc_dropout(enc_out, training = training) + target_norm)\n        ffn_out = self.ffn(enc_out_norm)\n        ffn_out_norm = self.layernorm3(enc_out_norm + self.ffn_dropout(ffn_out, training = training))\n        return ffn_out_norm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:27.188042Z","iopub.execute_input":"2025-05-25T17:12:27.188281Z","iopub.status.idle":"2025-05-25T17:12:27.208421Z","shell.execute_reply.started":"2025-05-25T17:12:27.188258Z","shell.execute_reply":"2025-05-25T17:12:27.207669Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TransformerEncoderStack(keras.layers.Layer):\n    def __init__(self, num_layers, num_hid, num_head, num_feed_forward):\n        super().__init__()\n        self.enc_layers = [\n            TransformerEncoder(num_hid, num_head, num_feed_forward)\n            for _ in range(num_layers)\n        ]\n\n    def call(self, x, training=False):\n        for layer in self.enc_layers:\n            x = layer(x, training=training)\n        return x\nclass CTCModel(keras.Model):\n    def __init__(\n        self,\n        num_hid=64,\n        num_head=2,\n        num_feed_forward=128,\n        source_maxlen=FRAME_LEN, # Độ dài đầu vào của LandmarkEmbedding\n        # Giá trị mặc định cho encoder_output_seqlen giờ đây là hằng số toàn cục đã được cập nhật\n        encoder_output_seqlen_param=ENCODER_OUTPUT_SEQ_LEN,\n        num_layers_enc=4,\n        num_ctc_classes=NUM_CTC_CLASSES,\n    ):\n        super().__init__()\n        self.loss_metric = keras.metrics.Mean(name=\"ctc_loss\")\n        self.edit_dist_metric = keras.metrics.Mean(name=\"edit_dist\")\n\n        # source_maxlen là đầu vào cho LandmarkEmbedding\n        self.landmark_embedding = LandmarkEmbedding(num_hid=num_hid, maxlen=source_maxlen)\n        # encoder_output_seqlen_param là đầu ra của LandmarkEmbedding, đầu vào của Positional Embedding và TransformerEncoder\n        self.pos_emb_encoder = layers.Embedding(input_dim=encoder_output_seqlen_param, output_dim=num_hid)\n        self.encoder_output_actual_seqlen = encoder_output_seqlen_param # Lưu lại để dùng trong train/test_step\n\n        self.encoder = TransformerEncoderStack(\n            num_layers=num_layers_enc,\n            num_hid=num_hid,\n            num_head=num_head,\n            num_feed_forward=num_feed_forward\n        )\n        self.classifier = layers.Dense(num_ctc_classes)\n\n    def call(self, source, training=False):\n        x = self.landmark_embedding(source) # Shape: (batch, self.encoder_output_actual_seqlen, num_hid)\n        batch_size = tf.shape(x)[0]\n        # encoder_seq_len nên được lấy từ shape của x sau landmark_embedding,\n        # hoặc tốt hơn là dùng self.encoder_output_actual_seqlen đã lưu\n        current_encoder_seq_len = tf.shape(x)[1] # Hoặc dùng self.encoder_output_actual_seqlen\n\n        # Đảm bảo positions được tạo với đúng độ dài\n        positions = tf.range(start=0, limit=current_encoder_seq_len, delta=1)\n        positions = tf.broadcast_to(positions, [batch_size, current_encoder_seq_len])\n        pos_embeddings = self.pos_emb_encoder(positions)\n        x = x + pos_embeddings\n\n        x_encoded = self.encoder(x, training=training)\n        logits = self.classifier(x_encoded) # Shape: (batch, self.encoder_output_actual_seqlen, num_ctc_classes)\n        return logits\n\n    @property\n    def metrics(self):\n        return [self.loss_metric, self.edit_dist_metric]\n\n    def ctc_batch_loss(self, y_true, y_pred, input_length, label_length):\n        y_pred_time_major = tf.transpose(y_pred, perm=[1, 0, 2])\n        loss = tf.nn.ctc_loss(\n            labels=y_true,\n            logits=y_pred_time_major,\n            label_length=label_length,\n            logit_length=input_length,\n            blank_index=BLANK_LABEL_IDX,\n            logits_time_major=True\n        )\n        return tf.reduce_mean(loss)\n\n    def _dense_to_sparse(self, dense_tensor, sequence_lengths):\n        mask = tf.sequence_mask(sequence_lengths, maxlen=tf.shape(dense_tensor)[1])\n        indices = tf.cast(tf.where(mask), tf.int64)\n        values = tf.cast(tf.boolean_mask(dense_tensor, mask), tf.int64)\n        dense_shape = tf.cast(tf.shape(dense_tensor), tf.int64)\n        return tf.SparseTensor(indices, values, dense_shape)\n\n    def train_step(self, batch):\n        source, target_labels, target_label_lengths = batch\n        batch_size_tf = tf.shape(source)[0]\n        \n        # ---- SỬ DỤNG self.encoder_output_actual_seqlen HOẶC THAM SỐ ĐÚNG ----\n        ctc_input_length = tf.fill([batch_size_tf], self.encoder_output_actual_seqlen)\n        # ---- KẾT THÚC SỬA ĐỔI ----\n        \n        ctc_input_length = tf.cast(ctc_input_length, tf.int32)\n        target_label_lengths_int32 = tf.cast(target_label_lengths, tf.int32)\n\n        with tf.GradientTape() as tape:\n            preds_logits = self(source, training=True)\n            loss = self.ctc_batch_loss(target_labels, preds_logits, ctc_input_length, target_label_lengths_int32)\n        \n        gradients = tape.gradient(loss, self.trainable_variables)\n        self.optimizer.apply_gradients(zip(gradients, self.trainable_variables))\n        self.loss_metric.update_state(loss)\n        \n        decoded_sparse_hyp, _ = tf.nn.ctc_greedy_decoder(\n            tf.transpose(preds_logits, perm=[1, 0, 2]),\n            ctc_input_length\n        )\n        sparse_true_truth = self._dense_to_sparse(target_labels, target_label_lengths_int32)\n        edit_dist = tf.edit_distance(decoded_sparse_hyp[0], sparse_true_truth, normalize=False)\n        self.edit_dist_metric.update_state(tf.reduce_mean(edit_dist))\n        return {\"ctc_loss\": self.loss_metric.result(), \"edit_dist\": self.edit_dist_metric.result()}\n\n    def test_step(self, batch):\n        source, target_labels, target_label_lengths = batch\n        batch_size_tf = tf.shape(source)[0]\n        \n        # ---- SỬ DỤNG self.encoder_output_actual_seqlen HOẶC THAM SỐ ĐÚNG ----\n        ctc_input_length = tf.fill([batch_size_tf], self.encoder_output_actual_seqlen)\n        # ---- KẾT THÚC SỬA ĐỔI ----\n        \n        ctc_input_length = tf.cast(ctc_input_length, tf.int32)\n        target_label_lengths_int32 = tf.cast(target_label_lengths, tf.int32)\n\n        preds_logits = self(source, training=False)\n        loss = self.ctc_batch_loss(target_labels, preds_logits, ctc_input_length, target_label_lengths_int32)\n        self.loss_metric.update_state(loss)\n\n        decoded_sparse_hyp, _ = tf.nn.ctc_greedy_decoder(\n            tf.transpose(preds_logits, perm=[1, 0, 2]), \n            ctc_input_length\n        )\n        sparse_true_truth = self._dense_to_sparse(target_labels, target_label_lengths_int32)\n        edit_dist = tf.edit_distance(decoded_sparse_hyp[0], sparse_true_truth, normalize=False)\n        self.edit_dist_metric.update_state(tf.reduce_mean(edit_dist))\n        return {\"ctc_loss\": self.loss_metric.result(), \"edit_dist\": self.edit_dist_metric.result()}\n\n    def generate(self, source):\n        preds_logits = self(source, training=False)\n        batch_size = tf.shape(source)[0]\n        \n        # ---- SỬ DỤNG self.encoder_output_actual_seqlen HOẶC THAM SỐ ĐÚNG ----\n        ctc_input_length = tf.fill([batch_size], self.encoder_output_actual_seqlen)\n        # ---- KẾT THÚC SỬA ĐỔI ----\n\n        ctc_input_length = tf.cast(ctc_input_length, tf.int32)\n        decoded_sparse, _ = tf.nn.ctc_greedy_decoder(\n            tf.transpose(preds_logits, perm=[1, 0, 2]),\n            sequence_length=ctc_input_length\n        )\n        decoded_dense = tf.sparse.to_dense(decoded_sparse[0], default_value=tf.cast(BLANK_LABEL_IDX, tf.int64))\n        return decoded_dense","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:27.20915Z","iopub.execute_input":"2025-05-25T17:12:27.209407Z","iopub.status.idle":"2025-05-25T17:12:27.228996Z","shell.execute_reply.started":"2025-05-25T17:12:27.209383Z","shell.execute_reply":"2025-05-25T17:12:27.228394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DisplayOutputs(keras.callbacks.Callback):\n    def __init__(self, batch_data, idx_to_token_map):\n        # batch_data is (source, target_labels, target_label_lengths)\n        self.batch_source = batch_data[0]\n        self.batch_target_labels = batch_data[1]\n        self.batch_target_lengths = batch_data[2] # Store target lengths\n        self.idx_to_char = idx_to_token_map\n        # self.target_start_token_idx is no longer needed for CTC generation\n\n    def on_epoch_end(self, epoch, logs=None):\n        if (epoch + 1) % 4 != 0:\n            return\n        \n        source_for_display = self.batch_source # Use the stored batch source\n        \n        # --- CORRECTED CALL TO GENERATE ---\n        preds_indices_dense = self.model.generate(source_for_display) # Now takes only source\n        # preds_indices_dense is a dense tensor (batch_size, max_decoded_len)\n        # Values are character indices, padded with BLANK_LABEL_IDX (e.g., 59)\n        # --- END CORRECTION ---\n\n        preds_indices_np = preds_indices_dense.numpy()\n\n        print(f\"\\n--- Epoch {epoch+1} Sample Predictions ---\")\n        # Use min(5, actual_batch_size_of_display_batch) for looping\n        num_samples_to_display = min(5, source_for_display.shape[0]) \n\n        for i in range(num_samples_to_display):\n            # Get true label\n            target_len = self.batch_target_lengths[i].numpy()\n            target_text_indices = self.batch_target_labels[i, :target_len].numpy()\n            target_text = \"\".join([self.idx_to_char.get(idx, '?') for idx in target_text_indices])\n            \n            # Get predicted label from dense tensor\n            prediction_row = preds_indices_np[i]\n            \n            # Filter out BLANK_LABEL_IDX (and potentially PAD_TOKEN_LABEL_VALUE if it somehow appears)\n            # to form the predicted string\n            predicted_chars = []\n            for idx in prediction_row:\n                if idx == BLANK_LABEL_IDX: # Stop if we hit padding from ctc_decoder to_dense\n                    break \n                char = self.idx_to_char.get(idx)\n                if char is not None: # Ensure char exists (handles case where an unexpected index appears)\n                    predicted_chars.append(char)\n            prediction_text = \"\".join(predicted_chars)\n\n            print(f\"Target:     {target_text}\")\n            print(f\"Prediction: {prediction_text}\\n\")\n        print(f\"--- End Epoch {epoch+1} Sample Predictions ---\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:27.229602Z","iopub.execute_input":"2025-05-25T17:12:27.22987Z","iopub.status.idle":"2025-05-25T17:12:27.247679Z","shell.execute_reply.started":"2025-05-25T17:12:27.229849Z","shell.execute_reply":"2025-05-25T17:12:27.24717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping\nimport tensorflow as tf \n\n# Prepare a batch for DisplayOutputs\nval_iter = iter(valid_ds)\ndisplay_batch_data = next(val_iter) \n\ndisplay_cb = DisplayOutputs(display_batch_data, num_to_char) \n\n# ---- MODIFICATION HERE ----\nearlystop_cb = EarlyStopping(\n    monitor='val_ctc_loss', \n    patience=5,            \n    restore_best_weights=True,\n    mode='min'  # Explicitly tell EarlyStopping to minimize this metric\n)\n# ---- END MODIFICATION ----\n\nctc_model = CTCModel(\n    num_hid=200,\n    num_head=4,\n    num_feed_forward=400,\n    source_maxlen=FRAME_LEN,\n    encoder_output_seqlen_param=ENCODER_OUTPUT_SEQ_LEN, # Truyền vào giá trị đã cập nhật\n    num_layers_enc=2,\n    num_ctc_classes=NUM_CTC_CLASSES\n)\n\noptimizer = tf.keras.optimizers.Adam(learning_rate=0.0001)\n\nctc_model.compile(optimizer=optimizer, jit_compile=False) \n\nhistory = ctc_model.fit(train_ds, \n                        validation_data=valid_ds, \n                        callbacks=[display_cb, earlystop_cb], \n                        epochs=100)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:12:27.248613Z","iopub.execute_input":"2025-05-25T17:12:27.248928Z","iopub.status.idle":"2025-05-25T17:15:10.720656Z","shell.execute_reply.started":"2025-05-25T17:12:27.248908Z","shell.execute_reply":"2025-05-25T17:15:10.719811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.plot(history.history['ctc_loss'])\nplt.plot(history.history['val_ctc_loss'])\nplt.title('CTC Loss vs. Validation CTC Loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(['Training CTC Loss', 'Validation CTC Loss'])\nplt.show()\n\nplt.plot(history.history['edit_dist'])\nplt.plot(history.history['val_edit_dist'])\nplt.title('Edit Distance vs. Validation Edit Distance')\nplt.ylabel('Edit Distance')\nplt.xlabel('Epoch')\nplt.legend(['Training Edit Distance', 'Validation Edit Distance'])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:15:10.721521Z","iopub.execute_input":"2025-05-25T17:15:10.721787Z","iopub.status.idle":"2025-05-25T17:15:11.411818Z","shell.execute_reply.started":"2025-05-25T17:15:10.721771Z","shell.execute_reply":"2025-05-25T17:15:11.410982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_df = pd.read_csv('/kaggle/input/asl-fingerspelling/supplemental_metadata.csv')\ntf_records = dataset_df.file_id.map(lambda x: f'/kaggle/input/test-fsp/test/{x}.tfrecord').unique()\nprint(f\"List of {len(tf_records)} TFRecord files.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:27:48.931515Z","iopub.execute_input":"2025-05-25T17:27:48.932087Z","iopub.status.idle":"2025-05-25T17:27:49.02695Z","shell.execute_reply.started":"2025-05-25T17:27:48.932062Z","shell.execute_reply":"2025-05-25T17:27:49.026012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ds = tf.data.TFRecordDataset(tf_records).map(decode_fn).map(convert_fn).batch(batch_size).prefetch(buffer_size=tf.data.AUTOTUNE).cache()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:27:50.865684Z","iopub.execute_input":"2025-05-25T17:27:50.865968Z","iopub.status.idle":"2025-05-25T17:27:51.375975Z","shell.execute_reply.started":"2025-05-25T17:27:50.865949Z","shell.execute_reply":"2025-05-25T17:27:51.375146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = ctc_model.evaluate(test_ds)\nprint(\"Loss:\", results[0])\nprint(\"Edit distance:\", results[1])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:27:53.194978Z","iopub.execute_input":"2025-05-25T17:27:53.195653Z","iopub.status.idle":"2025-05-25T17:29:27.946543Z","shell.execute_reply.started":"2025-05-25T17:27:53.19563Z","shell.execute_reply":"2025-05-25T17:29:27.945713Z"}},"outputs":[],"execution_count":null}]}