{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n!pip install mediapipe\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\"\"\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-07-23T16:01:31.240928Z","iopub.execute_input":"2023-07-23T16:01:31.24136Z","iopub.status.idle":"2023-07-23T16:01:49.580538Z","shell.execute_reply.started":"2023-07-23T16:01:31.241325Z","shell.execute_reply":"2023-07-23T16:01:49.579297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\nimport numpy as np\nimport pandas as pd\nimport pyarrow.parquet as pq\nimport tensorflow as tf\nimport json\nimport mediapipe\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport random\n\nfrom skimage.transform import resize\nfrom mediapipe.framework.formats import landmark_pb2\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tqdm.notebook import tqdm\nfrom tqdm import tqdm as TQ\nfrom matplotlib import animation, rc","metadata":{"execution":{"iopub.status.busy":"2023-07-23T16:01:49.58245Z","iopub.execute_input":"2023-07-23T16:01:49.582786Z","iopub.status.idle":"2023-07-23T16:02:01.970821Z","shell.execute_reply.started":"2023-07-23T16:01:49.582755Z","shell.execute_reply":"2023-07-23T16:02:01.969521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-07-23T16:02:01.97248Z","iopub.execute_input":"2023-07-23T16:02:01.973663Z","iopub.status.idle":"2023-07-23T16:02:02.162588Z","shell.execute_reply.started":"2023-07-23T16:02:01.973611Z","shell.execute_reply":"2023-07-23T16:02:02.161364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LPOSE = [13, 15, 17, 19, 21]\nRPOSE = [14, 16, 18, 20, 22]\nPOSE = LPOSE + RPOSE","metadata":{"execution":{"iopub.status.busy":"2023-07-23T16:02:02.165294Z","iopub.execute_input":"2023-07-23T16:02:02.165653Z","iopub.status.idle":"2023-07-23T16:02:02.171313Z","shell.execute_reply.started":"2023-07-23T16:02:02.165624Z","shell.execute_reply":"2023-07-23T16:02:02.169961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = [f'x_right_hand_{i}' for i in range(21)] + [f'x_left_hand_{i}' for i in range(21)] + [f'x_pose_{i}' for i in POSE]\nY = [f'y_right_hand_{i}' for i in range(21)] + [f'y_left_hand_{i}' for i in range(21)] + [f'y_pose_{i}' for i in POSE]\nZ = [f'z_right_hand_{i}' for i in range(21)] + [f'z_left_hand_{i}' for i in range(21)] + [f'z_pose_{i}' for i in POSE]\nFEATURE_COLUMNS = X + Y + Z","metadata":{"execution":{"iopub.status.busy":"2023-07-23T16:02:02.173241Z","iopub.execute_input":"2023-07-23T16:02:02.174158Z","iopub.status.idle":"2023-07-23T16:02:02.186439Z","shell.execute_reply.started":"2023-07-23T16:02:02.174114Z","shell.execute_reply":"2023-07-23T16:02:02.185213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"x_\" in col]\nY_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"y_\" in col]\nZ_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"z_\" in col]\n\nRHAND_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"right\" in col]\nLHAND_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"left\" in col]\nRPOSE_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"pose\" in col and int(col[-2:]) in RPOSE]\nLPOSE_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"pose\" in col and int(col[-2:]) in LPOSE]","metadata":{"execution":{"iopub.status.busy":"2023-07-23T16:02:02.187873Z","iopub.execute_input":"2023-07-23T16:02:02.188222Z","iopub.status.idle":"2023-07-23T16:02:02.208768Z","shell.execute_reply.started":"2023-07-23T16:02:02.188193Z","shell.execute_reply":"2023-07-23T16:02:02.207773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open (\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\", \"r\") as f:\n    char_to_num = json.load(f)\n\n# Add pad_token, start pointer and end pointer to the dict\npad_token = 'P'\nstart_token = '<'\nend_token = '>'\nLPOSE = [13, 15, 17, 19, 21]\nRPOSE = [14, 16, 18, 20, 22]\nPOSE = LPOSE + RPOSE\npad_token_idx = 59\nstart_token_idx = 60\nend_token_idx = 61\n\nchar_to_num[pad_token] = pad_token_idx\nchar_to_num[start_token] = start_token_idx\nchar_to_num[end_token] = end_token_idx\nnum_to_char = {j:i for i,j in char_to_num.items()}","metadata":{"execution":{"iopub.status.busy":"2023-07-23T16:02:02.210383Z","iopub.execute_input":"2023-07-23T16:02:02.211058Z","iopub.status.idle":"2023-07-23T16:02:02.226385Z","shell.execute_reply.started":"2023-07-23T16:02:02.211021Z","shell.execute_reply":"2023-07-23T16:02:02.225324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set length of frames to 128\nFRAME_LEN = 128\n\n# Function to resize and add padding.\ndef resize_pad(x):\n    if tf.shape(x)[0] < FRAME_LEN:\n        x = tf.pad(x, ([[0, FRAME_LEN-tf.shape(x)[0]], [0, 0], [0, 0]]))\n    else:\n        x = tf.image.resize(x, (FRAME_LEN, tf.shape(x)[1]))\n    return x\n\n# Detect the dominant hand from the number of NaN values.\n# Dominant hand will have less NaN values since it is in frame moving.\ndef pre_process(x):\n    rhand = tf.gather(x, RHAND_IDX, axis=1)\n    lhand = tf.gather(x, LHAND_IDX, axis=1)\n    rpose = tf.gather(x, RPOSE_IDX, axis=1)\n    lpose = tf.gather(x, LPOSE_IDX, axis=1)\n    \n    rnan_idx = tf.reduce_any(tf.math.is_nan(rhand), axis=1)\n    lnan_idx = tf.reduce_any(tf.math.is_nan(lhand), axis=1)\n    \n    rnans = tf.math.count_nonzero(rnan_idx)\n    lnans = tf.math.count_nonzero(lnan_idx)\n    \n    # For dominant hand\n    if rnans > lnans:\n        hand = lhand\n        pose = lpose\n        \n        hand_x = hand[:, 0*(len(LHAND_IDX)//3) : 1*(len(LHAND_IDX)//3)]\n        hand_y = hand[:, 1*(len(LHAND_IDX)//3) : 2*(len(LHAND_IDX)//3)]\n        hand_z = hand[:, 2*(len(LHAND_IDX)//3) : 3*(len(LHAND_IDX)//3)]\n        hand = tf.concat([1-hand_x, hand_y, hand_z], axis=1)\n        \n        pose_x = pose[:, 0*(len(LPOSE_IDX)//3) : 1*(len(LPOSE_IDX)//3)]\n        pose_y = pose[:, 1*(len(LPOSE_IDX)//3) : 2*(len(LPOSE_IDX)//3)]\n        pose_z = pose[:, 2*(len(LPOSE_IDX)//3) : 3*(len(LPOSE_IDX)//3)]\n        pose = tf.concat([1-pose_x, pose_y, pose_z], axis=1)\n    else:\n        hand = rhand\n        pose = rpose\n    \n        hand_x = hand[:, 0*(len(RHAND_IDX)//3) : 1*(len(RHAND_IDX)//3)]\n        hand_y = hand[:, 1*(len(RHAND_IDX)//3) : 2*(len(RHAND_IDX)//3)]\n        hand_z = hand[:, 2*(len(RHAND_IDX)//3) : 3*(len(RHAND_IDX)//3)]\n        hand = tf.concat([hand_x, hand_y, hand_z], axis=1)\n        \n        pose_x = pose[:, 0*(len(RPOSE_IDX)//3) : 1*(len(RPOSE_IDX)//3)]\n        pose_y = pose[:, 1*(len(RPOSE_IDX)//3) : 2*(len(RPOSE_IDX)//3)]\n        pose_z = pose[:, 2*(len(RPOSE_IDX)//3) : 3*(len(RPOSE_IDX)//3)]\n        pose = tf.concat([pose_x, pose_y, pose_z], axis=1)\n        \n        \n    hand = tf.concat([hand_x[..., tf.newaxis], hand_y[..., tf.newaxis], hand_z[..., tf.newaxis]], axis=-1)\n    \n    mean = tf.math.reduce_mean(hand, axis=1)[:, tf.newaxis, :]\n    std = tf.math.reduce_std(hand, axis=1)[:, tf.newaxis, :]\n    hand = (hand - mean) / std\n\n    pose = tf.concat([pose_x[..., tf.newaxis], pose_y[..., tf.newaxis], pose_z[..., tf.newaxis]], axis=-1)\n    \n    x = tf.concat([hand, pose], axis=1)\n    x = resize_pad(x)\n    \n    x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n    x = tf.reshape(x, (FRAME_LEN, len(LHAND_IDX) + len(LPOSE_IDX)))\n    return x\n","metadata":{"execution":{"iopub.status.busy":"2023-07-23T16:02:02.228434Z","iopub.execute_input":"2023-07-23T16:02:02.229022Z","iopub.status.idle":"2023-07-23T16:02:02.253583Z","shell.execute_reply.started":"2023-07-23T16:02:02.228988Z","shell.execute_reply":"2023-07-23T16:02:02.252266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# To Parse data from tfrecord","metadata":{}},{"cell_type":"code","source":"def decode_fn(record_bytes):\n    schema = {COL: tf.io.VarLenFeature(dtype=tf.float32) for COL in FEATURE_COLUMNS}\n    schema[\"phrase\"] = tf.io.FixedLenFeature([], dtype=tf.string)\n    features = tf.io.parse_single_example(record_bytes, schema)\n    phrase = features[\"phrase\"]\n    landmarks = ([tf.sparse.to_dense(features[COL]) for COL in FEATURE_COLUMNS])\n    # Transpose to maintain the original shape of landmarks data.\n    landmarks = tf.transpose(landmarks)\n    \n    return landmarks, phrase","metadata":{"execution":{"iopub.status.busy":"2023-07-23T16:02:02.255258Z","iopub.execute_input":"2023-07-23T16:02:02.25619Z","iopub.status.idle":"2023-07-23T16:02:02.274353Z","shell.execute_reply.started":"2023-07-23T16:02:02.256153Z","shell.execute_reply":"2023-07-23T16:02:02.273262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# To Convert Data","metadata":{}},{"cell_type":"code","source":"table = tf.lookup.StaticHashTable(\n    initializer=tf.lookup.KeyValueTensorInitializer(\n        keys=list(char_to_num.keys()),\n        values=list(char_to_num.values()),\n    ),\n    default_value=tf.constant(-1),\n    name=\"class_weight\"\n)\n\ndef convert_fn(landmarks, phrase):\n    # Add start and end pointers to phrase.\n    phrase = start_token + phrase + end_token\n    phrase = tf.strings.bytes_split(phrase)\n    phrase = table.lookup(phrase)\n    # Vectorize and add padding.\n    phrase = tf.pad(phrase, paddings=[[0, 64 - tf.shape(phrase)[0]]], mode = 'CONSTANT',\n                    constant_values = pad_token_idx)\n    # Apply pre_process function to the landmarks.\n    return pre_process(landmarks), phrase","metadata":{"execution":{"iopub.status.busy":"2023-07-23T16:02:02.277808Z","iopub.execute_input":"2023-07-23T16:02:02.278949Z","iopub.status.idle":"2023-07-23T16:02:02.404519Z","shell.execute_reply.started":"2023-07-23T16:02:02.278895Z","shell.execute_reply":"2023-07-23T16:02:02.403567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Map the tfrecord files into a variable","metadata":{}},{"cell_type":"code","source":"tf_records = dataset_df.file_id.map(lambda x: f'/kaggle/input/aslfr-parquets-to-tfrecords-cleaned/tfds/{x}.tfrecord').unique()","metadata":{"execution":{"iopub.status.busy":"2023-07-23T16:02:02.405954Z","iopub.execute_input":"2023-07-23T16:02:02.406622Z","iopub.status.idle":"2023-07-23T16:02:02.486083Z","shell.execute_reply.started":"2023-07-23T16:02:02.406588Z","shell.execute_reply":"2023-07-23T16:02:02.484859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 64\ntrain_len = int(0.8 * len(tf_records))\n\ntrain_ds = tf.data.TFRecordDataset(tf_records[:train_len]).map(decode_fn).map(convert_fn).batch(batch_size).prefetch(buffer_size=tf.data.AUTOTUNE).cache()\nvalid_ds = tf.data.TFRecordDataset(tf_records[train_len:]).map(decode_fn).map(convert_fn).batch(batch_size).prefetch(buffer_size=tf.data.AUTOTUNE).cache()","metadata":{"execution":{"iopub.status.busy":"2023-07-23T16:02:02.490362Z","iopub.execute_input":"2023-07-23T16:02:02.490775Z","iopub.status.idle":"2023-07-23T16:02:04.594904Z","shell.execute_reply.started":"2023-07-23T16:02:02.49074Z","shell.execute_reply":"2023-07-23T16:02:04.593746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds","metadata":{"execution":{"iopub.status.busy":"2023-07-23T16:03:27.498264Z","iopub.execute_input":"2023-07-23T16:03:27.498695Z","iopub.status.idle":"2023-07-23T16:03:27.506931Z","shell.execute_reply.started":"2023-07-23T16:03:27.498663Z","shell.execute_reply":"2023-07-23T16:03:27.505613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class LSTM_Encoder(layers.Layer):\n    def __init__(self, embed_dim, num_heads, feed_forward_dim, rate=0.1):\n        super.__init__()\n        self.enc_lstm_layer = keras.Sequential([\n            layers.LSTM(\n                4,\n                activation='tanh', \n                recurrent_activation='sigmoid',\n            ),\n            layers.Dense(embed_dim)    \n        ])\n        self.enc_layernorm = layers.LayerNormalization(epsilon=1e-6)\n\n    def call(self, inputs, training):\n        lstm_outputs = self.enc_lstm_layer()\n        return self.enc_layernorm(lstm_outputs)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}