{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":52950,"databundleVersionId":5973250,"sourceType":"competition"}],"dockerImageVersionId":30615,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install mediapipe","metadata":{"execution":{"iopub.status.busy":"2023-12-12T13:58:40.614503Z","iopub.execute_input":"2023-12-12T13:58:40.61695Z","iopub.status.idle":"2023-12-12T13:59:00.524708Z","shell.execute_reply.started":"2023-12-12T13:58:40.616898Z","shell.execute_reply":"2023-12-12T13:59:00.52305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\nimport numpy as np\nimport pandas as pd\nimport pyarrow.parquet as pq\nimport tensorflow as tf\nimport json\nimport mediapipe\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport random\n\nfrom skimage.transform import resize\nfrom mediapipe.framework.formats import landmark_pb2\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.losses import CategoricalCrossentropy\nfrom tqdm.notebook import tqdm\nfrom matplotlib import animation, rc\nfrom tensorflow.keras.layers import LSTM","metadata":{"execution":{"iopub.status.busy":"2023-12-12T13:59:00.528076Z","iopub.execute_input":"2023-12-12T13:59:00.528952Z","iopub.status.idle":"2023-12-12T13:59:17.276662Z","shell.execute_reply.started":"2023-12-12T13:59:00.528904Z","shell.execute_reply":"2023-12-12T13:59:17.275616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-12-12T13:59:17.278172Z","iopub.execute_input":"2023-12-12T13:59:17.279084Z","iopub.status.idle":"2023-12-12T13:59:17.469465Z","shell.execute_reply.started":"2023-12-12T13:59:17.27904Z","shell.execute_reply":"2023-12-12T13:59:17.468346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fetch sequence_id, file_id, phrase from first row\nsequence_id, file_id, phrase = dataset_df.iloc[1][['sequence_id', 'file_id', 'phrase']]\nprint(f\"sequence_id: {sequence_id}, file_id: {file_id}, phrase: {phrase}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-12T13:59:17.471252Z","iopub.execute_input":"2023-12-12T13:59:17.471966Z","iopub.status.idle":"2023-12-12T13:59:17.495055Z","shell.execute_reply.started":"2023-12-12T13:59:17.471927Z","shell.execute_reply":"2023-12-12T13:59:17.493644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fetch data from parquet file\nsample_sequence_df = pq.read_table(f\"/kaggle/input/asl-fingerspelling/train_landmarks/{str(file_id)}.parquet\",\n    filters=[[('sequence_id', '=', sequence_id)],]).to_pandas()\nprint(\"Full sequence dataset shape is {}\".format(sample_sequence_df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-12-12T13:59:17.499451Z","iopub.execute_input":"2023-12-12T13:59:17.500295Z","iopub.status.idle":"2023-12-12T13:59:21.878507Z","shell.execute_reply.started":"2023-12-12T13:59:17.500245Z","shell.execute_reply":"2023-12-12T13:59:21.877206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pose coordinates for hand movement.\nLPOSE = [13, 15, 17, 19, 21]\nRPOSE = [14, 16, 18, 20, 22]\nPOSE = LPOSE + RPOSE","metadata":{"execution":{"iopub.status.busy":"2023-12-12T13:59:21.880001Z","iopub.execute_input":"2023-12-12T13:59:21.880371Z","iopub.status.idle":"2023-12-12T13:59:21.88636Z","shell.execute_reply.started":"2023-12-12T13:59:21.880342Z","shell.execute_reply":"2023-12-12T13:59:21.885173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = [f'x_right_hand_{i}' for i in range(21)] + [f'x_left_hand_{i}' for i in range(21)] + [f'x_pose_{i}' for i in POSE]\nY = [f'y_right_hand_{i}' for i in range(21)] + [f'y_left_hand_{i}' for i in range(21)] + [f'y_pose_{i}' for i in POSE]\nZ = [f'z_right_hand_{i}' for i in range(21)] + [f'z_left_hand_{i}' for i in range(21)] + [f'z_pose_{i}' for i in POSE]","metadata":{"execution":{"iopub.status.busy":"2023-12-12T13:59:21.887604Z","iopub.execute_input":"2023-12-12T13:59:21.887924Z","iopub.status.idle":"2023-12-12T13:59:21.90028Z","shell.execute_reply.started":"2023-12-12T13:59:21.887897Z","shell.execute_reply":"2023-12-12T13:59:21.898753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURE_COLUMNS = X + Y + Z","metadata":{"execution":{"iopub.status.busy":"2023-12-12T13:59:21.901693Z","iopub.execute_input":"2023-12-12T13:59:21.902512Z","iopub.status.idle":"2023-12-12T13:59:21.912669Z","shell.execute_reply.started":"2023-12-12T13:59:21.902473Z","shell.execute_reply":"2023-12-12T13:59:21.91113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"x_\" in col]\nY_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"y_\" in col]\nZ_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"z_\" in col]\n\nRHAND_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"right\" in col]\nLHAND_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"left\" in col]\nRPOSE_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"pose\" in col and int(col[-2:]) in RPOSE]\nLPOSE_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"pose\" in col and int(col[-2:]) in LPOSE]","metadata":{"execution":{"iopub.status.busy":"2023-12-12T13:59:21.914129Z","iopub.execute_input":"2023-12-12T13:59:21.914539Z","iopub.status.idle":"2023-12-12T13:59:21.925796Z","shell.execute_reply.started":"2023-12-12T13:59:21.914507Z","shell.execute_reply":"2023-12-12T13:59:21.924436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set length of frames to 128\nFRAME_LEN = 128\n\n# Create directory to store the new data\nif not os.path.isdir(\"preprocessed\"):\n    os.mkdir(\"preprocessed\")\nelse:\n    shutil.rmtree(\"preprocessed\")\n    os.mkdir(\"preprocessed\")\n\n# Loop through each file_id\nfor file_id in tqdm(dataset_df.file_id.unique()):\n    # Parquet file name\n    pq_file = f\"/kaggle/input/asl-fingerspelling/train_landmarks/{file_id}.parquet\"\n    # Filter train.csv and fetch entries only for the relevant file_id\n    file_df = dataset_df.loc[dataset_df[\"file_id\"] == file_id]\n    # Fetch the parquet file\n    parquet_df = pq.read_table(f\"/kaggle/input/asl-fingerspelling/train_landmarks/{str(file_id)}.parquet\",\n                              columns=['sequence_id'] + FEATURE_COLUMNS).to_pandas()\n    # File name for the updated data\n    tf_file = f\"preprocessed/{file_id}.tfrecord\"\n    parquet_numpy = parquet_df.to_numpy()\n    # Initialize the pointer to write the output of \n    # each `for loop` below as a sequence into the file.\n    with tf.io.TFRecordWriter(tf_file) as file_writer:\n        # Loop through each sequence in file.\n        for seq_id, phrase in zip(file_df.sequence_id, file_df.phrase):\n            # Fetch sequence data\n            frames = parquet_numpy[parquet_df.index == seq_id]\n            \n            # Calculate the number of NaN values in each hand landmark\n            r_nonan = np.sum(np.sum(np.isnan(frames[:, RHAND_IDX]), axis = 1) == 0)\n            l_nonan = np.sum(np.sum(np.isnan(frames[:, LHAND_IDX]), axis = 1) == 0)\n            no_nan = max(r_nonan, l_nonan)\n            \n            if 2*len(phrase)<no_nan:\n                features = {FEATURE_COLUMNS[i]: tf.train.Feature(\n                    float_list=tf.train.FloatList(value=frames[:, i])) for i in range(len(FEATURE_COLUMNS))}\n                features[\"phrase\"] = tf.train.Feature(bytes_list=tf.train.BytesList(value=[bytes(phrase, 'utf-8')]))\n                record_bytes = tf.train.Example(features=tf.train.Features(feature=features)).SerializeToString()\n                file_writer.write(record_bytes)","metadata":{"execution":{"iopub.status.busy":"2023-12-12T13:59:21.927482Z","iopub.execute_input":"2023-12-12T13:59:21.929334Z","iopub.status.idle":"2023-12-12T14:12:22.201618Z","shell.execute_reply.started":"2023-12-12T13:59:21.9293Z","shell.execute_reply":"2023-12-12T14:12:22.199493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf_records = dataset_df.file_id.map(lambda x: f'/kaggle/working/preprocessed/{x}.tfrecord').unique()\nprint(f\"List of {len(tf_records)} TFRecord files.\")","metadata":{"execution":{"iopub.status.busy":"2023-12-12T14:12:22.204679Z","iopub.execute_input":"2023-12-12T14:12:22.205249Z","iopub.status.idle":"2023-12-12T14:12:22.285803Z","shell.execute_reply.started":"2023-12-12T14:12:22.205188Z","shell.execute_reply":"2023-12-12T14:12:22.284851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open (\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\", \"r\") as f:\n    char_to_num = json.load(f)\n\n# Add pad_token, start pointer and end pointer to the dict\npad_token = 'P'\nstart_token = '<'\nend_token = '>'\npad_token_idx = 59\nstart_token_idx = 60\nend_token_idx = 61\n\nchar_to_num[pad_token] = pad_token_idx\nchar_to_num[start_token] = start_token_idx\nchar_to_num[end_token] = end_token_idx\nnum_to_char = {j:i for i,j in char_to_num.items()}","metadata":{"execution":{"iopub.status.busy":"2023-12-12T17:12:49.056723Z","iopub.execute_input":"2023-12-12T17:12:49.058416Z","iopub.status.idle":"2023-12-12T17:12:49.076386Z","shell.execute_reply.started":"2023-12-12T17:12:49.058354Z","shell.execute_reply":"2023-12-12T17:12:49.074701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/code/irohith/aslfr-transformer/notebook\n\n# Function to resize and add padding.\ndef resize_pad(x):\n    if tf.shape(x)[0] < FRAME_LEN:\n        x = tf.pad(x, ([[0, FRAME_LEN-tf.shape(x)[0]], [0, 0], [0, 0]]))\n    else:\n        x = tf.image.resize(x, (FRAME_LEN, tf.shape(x)[1]))\n    return x\n\n# Detect the dominant hand from the number of NaN values.\n# Dominant hand will have less NaN values since it is in frame moving.\ndef pre_process(x):\n    rhand = tf.gather(x, RHAND_IDX, axis=1)\n    lhand = tf.gather(x, LHAND_IDX, axis=1)\n    rpose = tf.gather(x, RPOSE_IDX, axis=1)\n    lpose = tf.gather(x, LPOSE_IDX, axis=1)\n    \n    rnan_idx = tf.reduce_any(tf.math.is_nan(rhand), axis=1)\n    lnan_idx = tf.reduce_any(tf.math.is_nan(lhand), axis=1)\n    \n    rnans = tf.math.count_nonzero(rnan_idx)\n    lnans = tf.math.count_nonzero(lnan_idx)\n    \n    # For dominant hand\n    if rnans > lnans:\n        hand = lhand\n        pose = lpose\n        \n        hand_x = hand[:, 0*(len(LHAND_IDX)//3) : 1*(len(LHAND_IDX)//3)]\n        hand_y = hand[:, 1*(len(LHAND_IDX)//3) : 2*(len(LHAND_IDX)//3)]\n        hand_z = hand[:, 2*(len(LHAND_IDX)//3) : 3*(len(LHAND_IDX)//3)]\n        hand = tf.concat([1-hand_x, hand_y, hand_z], axis=1)\n        \n        pose_x = pose[:, 0*(len(LPOSE_IDX)//3) : 1*(len(LPOSE_IDX)//3)]\n        pose_y = pose[:, 1*(len(LPOSE_IDX)//3) : 2*(len(LPOSE_IDX)//3)]\n        pose_z = pose[:, 2*(len(LPOSE_IDX)//3) : 3*(len(LPOSE_IDX)//3)]\n        pose = tf.concat([1-pose_x, pose_y, pose_z], axis=1)\n    else:\n        hand = rhand\n        pose = rpose\n    \n    hand_x = hand[:, 0*(len(LHAND_IDX)//3) : 1*(len(LHAND_IDX)//3)]\n    hand_y = hand[:, 1*(len(LHAND_IDX)//3) : 2*(len(LHAND_IDX)//3)]\n    hand_z = hand[:, 2*(len(LHAND_IDX)//3) : 3*(len(LHAND_IDX)//3)]\n    hand = tf.concat([hand_x[..., tf.newaxis], hand_y[..., tf.newaxis], hand_z[..., tf.newaxis]], axis=-1)\n    \n    mean = tf.math.reduce_mean(hand, axis=1)[:, tf.newaxis, :]\n    std = tf.math.reduce_std(hand, axis=1)[:, tf.newaxis, :]\n    hand = (hand - mean) / std\n\n    pose_x = pose[:, 0*(len(LPOSE_IDX)//3) : 1*(len(LPOSE_IDX)//3)]\n    pose_y = pose[:, 1*(len(LPOSE_IDX)//3) : 2*(len(LPOSE_IDX)//3)]\n    pose_z = pose[:, 2*(len(LPOSE_IDX)//3) : 3*(len(LPOSE_IDX)//3)]\n    pose = tf.concat([pose_x[..., tf.newaxis], pose_y[..., tf.newaxis], pose_z[..., tf.newaxis]], axis=-1)\n    \n    x = tf.concat([hand, pose], axis=1)\n    x = resize_pad(x)\n    \n    x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n    #x = tf.reshape(x, (FRAME_LEN, len(LHAND_IDX) + len(LPOSE_IDX)))\n    x = tf.reshape(x, [FRAME_LEN, len(LHAND_IDX) + len(LPOSE_IDX)])\n    #landmarks = tf.reshape(landmarks, [-1, frames, rows, cols, 1]) \n    return x","metadata":{"execution":{"iopub.status.busy":"2023-12-12T17:12:53.020132Z","iopub.execute_input":"2023-12-12T17:12:53.020981Z","iopub.status.idle":"2023-12-12T17:12:53.044658Z","shell.execute_reply.started":"2023-12-12T17:12:53.02093Z","shell.execute_reply":"2023-12-12T17:12:53.043372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_fn(record_bytes):\n    schema = {COL: tf.io.VarLenFeature(dtype=tf.float32) for COL in FEATURE_COLUMNS}\n    schema[\"phrase\"] = tf.io.FixedLenFeature([], dtype=tf.string)\n    features = tf.io.parse_single_example(record_bytes, schema)\n    phrase = features[\"phrase\"]\n    landmarks = ([tf.sparse.to_dense(features[COL]) for COL in FEATURE_COLUMNS])\n    # Transpose to maintain the original shape of landmarks data.\n    landmarks = tf.transpose(landmarks)\n    \n    return landmarks, phrase","metadata":{"execution":{"iopub.status.busy":"2023-12-12T17:12:57.160812Z","iopub.execute_input":"2023-12-12T17:12:57.161262Z","iopub.status.idle":"2023-12-12T17:12:57.169305Z","shell.execute_reply.started":"2023-12-12T17:12:57.161202Z","shell.execute_reply":"2023-12-12T17:12:57.167893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"table = tf.lookup.StaticHashTable(\n    initializer=tf.lookup.KeyValueTensorInitializer(\n        keys=list(char_to_num.keys()),\n        values=list(char_to_num.values()),\n    ),\n    default_value=tf.constant(-1),\n    name=\"class_weight\"\n)\n\ndef convert_fn(landmarks, phrase):\n    # Add start and end pointers to phrase.\n    phrase = start_token + phrase + end_token\n    phrase = tf.strings.bytes_split(phrase)\n    phrase = table.lookup(phrase)\n    # Vectorize and add padding.\n    phrase = tf.pad(phrase, paddings=[[0, 64 - tf.shape(phrase)[0]]], mode = 'CONSTANT',\n                    constant_values = pad_token_idx)\n    # Apply pre_process function to the landmarks.\n    return pre_process(landmarks), phrase","metadata":{"execution":{"iopub.status.busy":"2023-12-12T17:13:16.683897Z","iopub.execute_input":"2023-12-12T17:13:16.684372Z","iopub.status.idle":"2023-12-12T17:13:16.700059Z","shell.execute_reply.started":"2023-12-12T17:13:16.684336Z","shell.execute_reply":"2023-12-12T17:13:16.698718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 64\ntrain_len = int(0.8 * len(tf_records))\n\ntrain_ds = tf.data.TFRecordDataset(tf_records[:train_len]).map(decode_fn).map(convert_fn).batch(batch_size).prefetch(buffer_size=tf.data.AUTOTUNE).cache()\nvalid_ds = tf.data.TFRecordDataset(tf_records[train_len:]).map(decode_fn).map(convert_fn).batch(batch_size).prefetch(buffer_size=tf.data.AUTOTUNE).cache()","metadata":{"execution":{"iopub.status.busy":"2023-12-12T17:13:19.074694Z","iopub.execute_input":"2023-12-12T17:13:19.075188Z","iopub.status.idle":"2023-12-12T17:13:20.919136Z","shell.execute_reply.started":"2023-12-12T17:13:19.075145Z","shell.execute_reply":"2023-12-12T17:13:20.917777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Xây dựng mô hình CNN\nmodel = keras.Sequential(name='asl_cnn')\n\n# Lớp Conv2D đầu tiên \nmodel.add(layers.Conv2D(32, (3, 3), activation='relu', input_shape=(FRAME_LEN, len(LHAND_IDX) + len(LPOSE_IDX), 1))) \n\n# Reshape về 4 chiều  \n#model.add(layers.Flatten())\nmodel.add(layers.Reshape((126, 76, 32)))\n\n#print(layers)\n\n#model.add(layers.Reshape((128, 76, 154, 32)))\n\nmodel.add(layers.MaxPooling2D((2, 2)))\n\n# Lớp Conv2D thứ 2\nmodel.add(layers.Conv2D(64, (3, 3), activation='relu'))\n\n# Reshape về 4 chiều\n#model.add(layers.Flatten())\nmodel.add(layers.Reshape((61, 36, 64)))\n\n#model.add(layers.Reshape((126, 76, 64)))\n\nmodel.add(layers.MaxPooling2D((2, 2)))\n\n# Flatten véc-tơ đặc trưng \nmodel.add(layers.Flatten())  \n\n# Dense layer phân loại 59 ký tự đầu ra\nmodel.add(layers.Dense(64, activation='softmax'))\n\n# Chỉnh sửa pre_process để reshape input\n#def pre_process(landmarks):\n    # Các bước xử lý ban đầu\n    \n    # Reshape lại input\n    #landmarks = tf.reshape(landmarks, [-1, frames, rows, cols, 1]) \n    \n    #return landmarks\n\n# Huấn luyện mô hình\nmodel.compile(optimizer=Adam(), loss=CategoricalCrossentropy())\nmodel.fit(train_ds, validation_data=valid_ds, epochs=10)\n\n#model.save('asl_cnn_model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-12-12T17:13:30.460266Z","iopub.execute_input":"2023-12-12T17:13:30.460765Z","iopub.status.idle":"2023-12-12T18:08:53.359141Z","shell.execute_reply.started":"2023-12-12T17:13:30.460726Z","shell.execute_reply":"2023-12-12T18:08:53.357898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\ntest_loss, test_accuracy = model.evaluate(train_ds)\nprint(f'Test Accuracy: {test_accuracy}')\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-12-12T18:14:16.512154Z","iopub.execute_input":"2023-12-12T18:14:16.513322Z","iopub.status.idle":"2023-12-12T18:16:38.476127Z","shell.execute_reply.started":"2023-12-12T18:14:16.513279Z","shell.execute_reply":"2023-12-12T18:16:38.474797Z"},"trusted":true},"execution_count":null,"outputs":[]}]}