{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importações","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-24T05:02:47.53512Z","iopub.execute_input":"2023-08-24T05:02:47.535508Z","iopub.status.idle":"2023-08-24T05:02:47.550395Z","shell.execute_reply.started":"2023-08-24T05:02:47.535476Z","shell.execute_reply":"2023-08-24T05:02:47.54915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install mediapipe","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:02:47.55372Z","iopub.execute_input":"2023-08-24T05:02:47.554387Z","iopub.status.idle":"2023-08-24T05:02:59.580067Z","shell.execute_reply.started":"2023-08-24T05:02:47.554352Z","shell.execute_reply":"2023-08-24T05:02:59.578594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\nimport numpy as np\nimport pandas as pd\nimport pyarrow.parquet as pq\nimport tensorflow as tf\nimport json\nimport mediapipe\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport random\n\nfrom skimage.transform import resize\nfrom mediapipe.framework.formats import landmark_pb2\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tqdm.notebook import tqdm\nfrom matplotlib import animation, rc","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:02:59.587455Z","iopub.execute_input":"2023-08-24T05:02:59.588278Z","iopub.status.idle":"2023-08-24T05:02:59.607043Z","shell.execute_reply.started":"2023-08-24T05:02:59.588229Z","shell.execute_reply":"2023-08-24T05:02:59.605896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"TensorFlow v\" + tf.__version__)\nprint(\"Mediapipe v\" + mediapipe.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:02:59.611461Z","iopub.execute_input":"2023-08-24T05:02:59.614264Z","iopub.status.idle":"2023-08-24T05:02:59.624131Z","shell.execute_reply.started":"2023-08-24T05:02:59.614227Z","shell.execute_reply":"2023-08-24T05:02:59.623084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data","metadata":{}},{"cell_type":"markdown","source":"## Funções para conversão de arquivos em algo (pode deixar melhor, dentro de uma classe talvez)","metadata":{}},{"cell_type":"code","source":"def convert_to_tf_record(metadata):\n    # Create directory to store the new data\n    if not os.path.isdir(\"preprocessed\"):\n        os.mkdir(\"preprocessed\")\n    else:\n        shutil.rmtree(\"preprocessed\")\n        os.mkdir(\"preprocessed\")\n\n    # Loop through each file_id\n    for file_id in tqdm(metadata.file_id.unique()):\n        # Parquet file name\n        pq_file = f\"/kaggle/input/asl-fingerspelling/train_landmarks/{file_id}.parquet\"\n        # Filter train.csv and fetch entries only for the relevant file_id\n        file_df = metadata.loc[metadata[\"file_id\"] == file_id]\n        # Fetch the parquet file\n        parquet_df = pq.read_table(f\"/kaggle/input/asl-fingerspelling/train_landmarks/{str(file_id)}.parquet\",\n                                  columns=['sequence_id'] + FEATURE_COLUMNS).to_pandas()\n        # File name for the updated data\n        tf_file = f\"preprocessed/{file_id}.tfrecord\"\n        parquet_numpy = parquet_df.to_numpy()\n        # Initialize the pointer to write the output of \n        # each `for loop` below as a sequence into the file.\n        with tf.io.TFRecordWriter(tf_file) as file_writer:\n            # Loop through each sequence in file.\n            for seq_id, phrase in zip(file_df.sequence_id, file_df.phrase):\n                # Fetch sequence data\n                frames = parquet_numpy[parquet_df.index == seq_id]\n\n                # Calculate the number of NaN values in each hand landmark\n                r_nonan = np.sum(np.sum(np.isnan(frames[:, RHAND_IDX]), axis = 1) == 0)\n                l_nonan = np.sum(np.sum(np.isnan(frames[:, LHAND_IDX]), axis = 1) == 0)\n                no_nan = max(r_nonan, l_nonan)\n\n                if 2*len(phrase)<no_nan:\n                    features = {FEATURE_COLUMNS[i]: tf.train.Feature(\n                        float_list=tf.train.FloatList(value=frames[:, i])) for i in range(len(FEATURE_COLUMNS))}\n                    features[\"phrase\"] = tf.train.Feature(bytes_list=tf.train.BytesList(value=[bytes(phrase, 'utf-8')]))\n                    record_bytes = tf.train.Example(features=tf.train.Features(feature=features)).SerializeToString()\n                    file_writer.write(record_bytes)","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:02:59.630346Z","iopub.execute_input":"2023-08-24T05:02:59.631071Z","iopub.status.idle":"2023-08-24T05:02:59.65128Z","shell.execute_reply.started":"2023-08-24T05:02:59.631037Z","shell.execute_reply":"2023-08-24T05:02:59.650256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"This json file contains a character and its value. We will add three new characters, \"<\" and \">\"\nto mark the start and end of each phrase, and \"P\" for padding.\"\"\"\ndef get_char_to_num():\n    with open (\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\", \"r\") as f:\n        char_to_num = json.load(f)\n\n    char_to_num[pad_token] = pad_token_idx\n    char_to_num[start_token] = start_token_idx\n    char_to_num[end_token] = end_token_idx\n    #num_to_char = {j:i for i,j in char_to_num.items()} # isso aqui nem é usado\n    return char_to_num\n","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:02:59.656342Z","iopub.execute_input":"2023-08-24T05:02:59.658918Z","iopub.status.idle":"2023-08-24T05:02:59.667478Z","shell.execute_reply.started":"2023-08-24T05:02:59.658882Z","shell.execute_reply":"2023-08-24T05:02:59.666266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Variáveis que são utilizadas entre as classes","metadata":{}},{"cell_type":"code","source":"# Pose coordinates for hand movement.\nLPOSE = [13, 15, 17, 19, 21]\nRPOSE = [14, 16, 18, 20, 22]\nPOSE = LPOSE + RPOSE\n\n#Create x,y,z label names from coordinates\nX = [f'x_right_hand_{i}' for i in range(21)] + [f'x_left_hand_{i}' for i in range(21)] + [f'x_pose_{i}' for i in POSE]\nY = [f'y_right_hand_{i}' for i in range(21)] + [f'y_left_hand_{i}' for i in range(21)] + [f'y_pose_{i}' for i in POSE]\nZ = [f'z_right_hand_{i}' for i in range(21)] + [f'z_left_hand_{i}' for i in range(21)] + [f'z_pose_{i}' for i in POSE]\n\n#Create feature columns from the extracted coordinates.\nFEATURE_COLUMNS = X + Y + Z\n\n#Store ids of each coordinate labels to lists\nX_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"x_\" in col]\nY_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"y_\" in col]\nZ_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"z_\" in col]\n\nRHAND_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if \"right\" in col]\nLHAND_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"left\" in col]\nRPOSE_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"pose\" in col and int(col[-2:]) in RPOSE]\nLPOSE_IDX = [i for i, col in enumerate(FEATURE_COLUMNS)  if  \"pose\" in col and int(col[-2:]) in LPOSE]\n\n\n# Set length of frames to 128\nFRAME_LEN = 128\n\n\n# Add pad_token, start pointer and end pointer to the dict\npad_token = 'P'\nstart_token = '<'\nend_token = '>'\npad_token_idx = 59\nstart_token_idx = 60\nend_token_idx = 61\n\n#Load character_to_prediction json file\nchar_to_num = get_char_to_num()\n\ntable = tf.lookup.StaticHashTable(\n    initializer=tf.lookup.KeyValueTensorInitializer(\n        keys=list(char_to_num.keys()),\n        values=list(char_to_num.values()),\n    ),\n    default_value=tf.constant(-1),\n    name=\"class_weight\"\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:02:59.672055Z","iopub.execute_input":"2023-08-24T05:02:59.674935Z","iopub.status.idle":"2023-08-24T05:02:59.702444Z","shell.execute_reply.started":"2023-08-24T05:02:59.674892Z","shell.execute_reply":"2023-08-24T05:02:59.701304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Faz pre processamento da dataset já convertido em tf_record","metadata":{}},{"cell_type":"code","source":"class DataProcessor:\n    # Function to resize and add padding.\n    @staticmethod\n    def resize_pad(x):\n        if tf.shape(x)[0] < FRAME_LEN:\n            x = tf.pad(x, ([[0, FRAME_LEN-tf.shape(x)[0]], [0, 0], [0, 0]]))\n        else:\n            x = tf.image.resize(x, (FRAME_LEN, tf.shape(x)[1]))\n        return x\n\n    # Detect the dominant hand from the number of NaN values.\n    # Dominant hand will have less NaN values since it is in frame moving.\n    @staticmethod\n    def pre_process(x):\n        rhand = tf.gather(x, RHAND_IDX, axis=1)\n        lhand = tf.gather(x, LHAND_IDX, axis=1)\n        rpose = tf.gather(x, RPOSE_IDX, axis=1)\n        lpose = tf.gather(x, LPOSE_IDX, axis=1)\n\n        rnan_idx = tf.reduce_any(tf.math.is_nan(rhand), axis=1)\n        lnan_idx = tf.reduce_any(tf.math.is_nan(lhand), axis=1)\n\n        rnans = tf.math.count_nonzero(rnan_idx)\n        lnans = tf.math.count_nonzero(lnan_idx)\n\n        # For dominant hand\n        if rnans > lnans:\n            hand = lhand\n            pose = lpose\n\n            hand_x = hand[:, 0*(len(LHAND_IDX)//3) : 1*(len(LHAND_IDX)//3)]\n            hand_y = hand[:, 1*(len(LHAND_IDX)//3) : 2*(len(LHAND_IDX)//3)]\n            hand_z = hand[:, 2*(len(LHAND_IDX)//3) : 3*(len(LHAND_IDX)//3)]\n            hand = tf.concat([1-hand_x, hand_y, hand_z], axis=1)\n\n            pose_x = pose[:, 0*(len(LPOSE_IDX)//3) : 1*(len(LPOSE_IDX)//3)]\n            pose_y = pose[:, 1*(len(LPOSE_IDX)//3) : 2*(len(LPOSE_IDX)//3)]\n            pose_z = pose[:, 2*(len(LPOSE_IDX)//3) : 3*(len(LPOSE_IDX)//3)]\n            pose = tf.concat([1-pose_x, pose_y, pose_z], axis=1)\n        else:\n            hand = rhand\n            pose = rpose\n\n        hand_x = hand[:, 0*(len(LHAND_IDX)//3) : 1*(len(LHAND_IDX)//3)]\n        hand_y = hand[:, 1*(len(LHAND_IDX)//3) : 2*(len(LHAND_IDX)//3)]\n        hand_z = hand[:, 2*(len(LHAND_IDX)//3) : 3*(len(LHAND_IDX)//3)]\n        hand = tf.concat([hand_x[..., tf.newaxis], hand_y[..., tf.newaxis], hand_z[..., tf.newaxis]], axis=-1)\n\n        mean = tf.math.reduce_mean(hand, axis=1)[:, tf.newaxis, :]\n        std = tf.math.reduce_std(hand, axis=1)[:, tf.newaxis, :]\n        hand = (hand - mean) / std\n\n        pose_x = pose[:, 0*(len(LPOSE_IDX)//3) : 1*(len(LPOSE_IDX)//3)]\n        pose_y = pose[:, 1*(len(LPOSE_IDX)//3) : 2*(len(LPOSE_IDX)//3)]\n        pose_z = pose[:, 2*(len(LPOSE_IDX)//3) : 3*(len(LPOSE_IDX)//3)]\n        pose = tf.concat([pose_x[..., tf.newaxis], pose_y[..., tf.newaxis], pose_z[..., tf.newaxis]], axis=-1)\n\n        x = tf.concat([hand, pose], axis=1)\n        x = DataProcessor.resize_pad(x)\n\n        x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n        x = tf.reshape(x, (FRAME_LEN, len(LHAND_IDX) + len(LPOSE_IDX)))\n        return x\n\n\n    \"\"\"This function will read the TFRecord data and convert it to Tensors.\"\"\"\n    @staticmethod\n    def decode_fn(record_bytes):\n        schema = {COL: tf.io.VarLenFeature(dtype=tf.float32) for COL in FEATURE_COLUMNS}\n        schema[\"phrase\"] = tf.io.FixedLenFeature([], dtype=tf.string)\n        features = tf.io.parse_single_example(record_bytes, schema)\n        phrase = features[\"phrase\"]\n        landmarks = ([tf.sparse.to_dense(features[COL]) for COL in FEATURE_COLUMNS])\n        # Transpose to maintain the original shape of landmarks data.\n        landmarks = tf.transpose(landmarks)\n\n        return landmarks, phrase\n\n\n\n    \"\"\"This function transposes and applies masks to the landmark coordinates. It also vectorizes the phrase corresponding to the landmarks using character_to_prediction_index.json.\"\"\"\n    @staticmethod\n    def convert_fn(landmarks, phrase):\n        # Add start and end pointers to phrase.\n        phrase = start_token + phrase + end_token\n        phrase = tf.strings.bytes_split(phrase)\n        phrase = table.lookup(phrase)\n        # Vectorize and add padding.\n        phrase = tf.pad(phrase, paddings=[[0, 64 - tf.shape(phrase)[0]]], mode = 'CONSTANT',\n                        constant_values = pad_token_idx)\n        # Apply pre_process function to the landmarks.\n        return DataProcessor.pre_process(landmarks), phrase","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:02:59.707414Z","iopub.execute_input":"2023-08-24T05:02:59.709589Z","iopub.status.idle":"2023-08-24T05:02:59.746456Z","shell.execute_reply.started":"2023-08-24T05:02:59.709548Z","shell.execute_reply":"2023-08-24T05:02:59.745559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Data:\n    def __init__(self):\n        self.metadata = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\n        self.batch_size = 64\n        \n        # se não existir cria\n        if not os.path.exists('/kaggle/working/preprocessed'):\n            os.makedirs('/kaggle/working/preprocessed')\n        # Se tiver vazio preenche\n        if not os.listdir('/kaggle/working/preprocessed'):\n            convert_to_tf_record(self.metadata)\n            \n        self.tf_records = self.metadata.file_id.map(lambda x: f'/kaggle/working/preprocessed/{x}.tfrecord').unique()\n        \n        self.train_size = 0.8\n        self.train_len = int(self.lenght * self.train_size)\n    \n    def get_sample(self, sequence_id, file_id):\n        sample_sequence = pq.read_table(f\"/kaggle/input/asl-fingerspelling/train_landmarks/{str(file_id)}.parquet\",filters=[[('sequence_id', '=', sequence_id)],]).to_pandas()\n        if len(sample_sequence) == 0:\n                    raise Exception(\"Sample vazia ou não encontrada!\")\n        return sample_sequence\n    \n    #lenght de tf_records!  detalhe\n    @property\n    def lenght(self):\n        return int(len(self.tf_records))\n\n    \n    def get_dataloader(self,train = True):\n        if train:\n            train_ds = tf.data.TFRecordDataset(self.tf_records[:self.train_len]).map(DataProcessor.decode_fn).map(DataProcessor.convert_fn).batch(self.batch_size).prefetch(buffer_size=tf.data.AUTOTUNE).cache()\n            return train_ds\n        else:\n            valid_ds = tf.data.TFRecordDataset(self.tf_records[self.train_len:]).map(DataProcessor.decode_fn).map(DataProcessor.convert_fn).batch(self.batch_size).prefetch(buffer_size=tf.data.AUTOTUNE).cache()\n            return valid_ds\n        ","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:02:59.752022Z","iopub.execute_input":"2023-08-24T05:02:59.754805Z","iopub.status.idle":"2023-08-24T05:02:59.769969Z","shell.execute_reply.started":"2023-08-24T05:02:59.754771Z","shell.execute_reply":"2023-08-24T05:02:59.768905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_data = Data()","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:02:59.77456Z","iopub.execute_input":"2023-08-24T05:02:59.777351Z","iopub.status.idle":"2023-08-24T05:03:00.004063Z","shell.execute_reply.started":"2023-08-24T05:02:59.777269Z","shell.execute_reply":"2023-08-24T05:03:00.003013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_ds = my_data.get_dataloader(train=False)\ntrain_ds = my_data.get_dataloader()","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:00.012353Z","iopub.execute_input":"2023-08-24T05:03:00.014631Z","iopub.status.idle":"2023-08-24T05:03:01.228673Z","shell.execute_reply.started":"2023-08-24T05:03:00.014594Z","shell.execute_reply":"2023-08-24T05:03:01.227575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch = next(iter(valid_ds))","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:01.234021Z","iopub.execute_input":"2023-08-24T05:03:01.23634Z","iopub.status.idle":"2023-08-24T05:03:01.665891Z","shell.execute_reply.started":"2023-08-24T05:03:01.236304Z","shell.execute_reply":"2023-08-24T05:03:01.664509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# landmark\nbatch[0].shape","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:01.670707Z","iopub.execute_input":"2023-08-24T05:03:01.672963Z","iopub.status.idle":"2023-08-24T05:03:01.688407Z","shell.execute_reply.started":"2023-08-24T05:03:01.672915Z","shell.execute_reply":"2023-08-24T05:03:01.682326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# phrase\nbatch[1].shape","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:01.691065Z","iopub.execute_input":"2023-08-24T05:03:01.6915Z","iopub.status.idle":"2023-08-24T05:03:01.719919Z","shell.execute_reply.started":"2023-08-24T05:03:01.691458Z","shell.execute_reply":"2023-08-24T05:03:01.717446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Rede","metadata":{}},{"cell_type":"code","source":"\"\"\"When processing landmark coordinate features for the encoder, we apply convolutional layers to downsample them and process local relationships.\n\nWe sum position embeddings and token embeddings when processing past target tokens for the decoder.\"\"\"\n\nclass TokenEmbedding(layers.Layer):\n    def __init__(self, num_vocab=1000, maxlen=100, num_hid=64):\n        super().__init__()\n        self.emb = tf.keras.layers.Embedding(num_vocab, num_hid)\n        self.pos_emb = layers.Embedding(input_dim=maxlen, output_dim=num_hid)\n\n    def call(self, x):\n        maxlen = tf.shape(x)[-1]\n        x = self.emb(x)\n        positions = tf.range(start=0, limit=maxlen, delta=1)\n        positions = self.pos_emb(positions)\n        return x + positions\n\n\nclass LandmarkEmbedding(layers.Layer):\n    def __init__(self, num_hid=64, maxlen=100):\n        super().__init__()\n        self.conv1 = tf.keras.layers.Conv1D(\n            num_hid, 11, strides=2, padding=\"same\", activation=\"relu\"\n        )\n        self.conv2 = tf.keras.layers.Conv1D(\n            num_hid, 11, strides=2, padding=\"same\", activation=\"relu\"\n        )\n        self.conv3 = tf.keras.layers.Conv1D(\n            num_hid, 11, strides=2, padding=\"same\", activation=\"relu\"\n        )\n        self.pos_emb = layers.Embedding(input_dim=maxlen, output_dim=num_hid)\n\n    def call(self, x):\n        x = self.conv1(x)\n        x = self.conv2(x)\n        return self.conv3(x)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:01.721871Z","iopub.execute_input":"2023-08-24T05:03:01.722373Z","iopub.status.idle":"2023-08-24T05:03:01.742068Z","shell.execute_reply.started":"2023-08-24T05:03:01.722334Z","shell.execute_reply":"2023-08-24T05:03:01.740942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TransformerEncoder(layers.Layer):\n    def __init__(self, embed_dim, num_heads, feed_forward_dim, rate=0.1):\n        super().__init__()\n        self.att = layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        self.ffn = keras.Sequential(\n            [\n                layers.Dense(feed_forward_dim, activation=\"relu\"),\n                layers.Dense(embed_dim),\n            ]\n        )\n        self.layernorm1 = layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm2 = layers.LayerNormalization(epsilon=1e-6)\n        self.dropout1 = layers.Dropout(rate)\n        self.dropout2 = layers.Dropout(rate)\n\n    def call(self, inputs, training):\n        attn_output = self.att(inputs, inputs)\n        attn_output = self.dropout1(attn_output, training=training)\n        out1 = self.layernorm1(inputs + attn_output)\n        ffn_output = self.ffn(out1)\n        ffn_output = self.dropout2(ffn_output, training=training)\n        return self.layernorm2(out1 + ffn_output)\n    \n\n# Customized to add `training` variable\n# Reference: https://www.kaggle.com/code/shlomoron/aslfr-a-simple-transformer/notebook\n\nclass TransformerDecoder(layers.Layer):\n    def __init__(self, embed_dim, num_heads, feed_forward_dim, dropout_rate=0.1):\n        super().__init__()\n        self.layernorm1 = layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm2 = layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm3 = layers.LayerNormalization(epsilon=1e-6)\n        \n        self.self_att = layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        self.enc_att = layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        \n        self.self_dropout = layers.Dropout(0.5)\n        self.enc_dropout = layers.Dropout(0.1)\n        self.ffn_dropout = layers.Dropout(0.1)\n        \n        self.ffn = keras.Sequential(\n            [\n                layers.Dense(feed_forward_dim, activation=\"relu\"),\n                layers.Dense(embed_dim),\n            ]\n        )\n\n    def causal_attention_mask(self, batch_size, n_dest, n_src, dtype):\n        \"\"\"Masks the upper half of the dot product matrix in self attention.\n\n        This prevents flow of information from future tokens to current token.\n        1's in the lower triangle, counting from the lower right corner.\n        \"\"\"\n        i = tf.range(n_dest)[:, None]\n        j = tf.range(n_src)\n        m = i >= j - n_src + n_dest\n        mask = tf.cast(m, dtype)\n        mask = tf.reshape(mask, [1, n_dest, n_src])\n        mult = tf.concat(\n            [batch_size[..., tf.newaxis], tf.constant([1, 1], dtype=tf.int32)], 0\n        )\n        return tf.tile(mask, mult)\n\n    def call(self, enc_out, target, training):\n        input_shape = tf.shape(target)\n        batch_size = input_shape[0]\n        seq_len = input_shape[1]\n        causal_mask = self.causal_attention_mask(batch_size, seq_len, seq_len, tf.bool)\n        target_att = self.self_att(target, target, attention_mask=causal_mask)\n        target_norm = self.layernorm1(target + self.self_dropout(target_att, training = training))\n        enc_out = self.enc_att(target_norm, enc_out)\n        enc_out_norm = self.layernorm2(self.enc_dropout(enc_out, training = training) + target_norm)\n        ffn_out = self.ffn(enc_out_norm)\n        ffn_out_norm = self.layernorm3(enc_out_norm + self.ffn_dropout(ffn_out, training = training))\n        return ffn_out_norm","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:01.745792Z","iopub.execute_input":"2023-08-24T05:03:01.746253Z","iopub.status.idle":"2023-08-24T05:03:01.777178Z","shell.execute_reply.started":"2023-08-24T05:03:01.746184Z","shell.execute_reply":"2023-08-24T05:03:01.775899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"This model takes landmark coordinates as inputs and predicts a sequence of characters. The target character sequence, which has been shifted to the left is provided as the input\nto the decoder during training. The decoder employs its own past predictions during inference to forecast the next token.\"\"\"\n\n\"\"\"The Levenshtein Distance between sequences is used as the accuracy metric since \nthe evaluation metric for this contest is the Normalized Total Levenshtein Distance.\n\"\"\"\nclass Transformer(keras.Model):\n    def __init__(\n        self,\n            num_hid=200,\n            num_head=4,\n            num_feed_forward=400,\n            source_maxlen = FRAME_LEN,\n            target_maxlen=64,\n            num_layers_enc=2,\n            num_layers_dec=1,\n            num_classes=62,\n    ):\n        \n        super().__init__()\n        self.num_layers_enc = num_layers_enc\n        self.num_layers_dec = num_layers_dec\n        self.target_maxlen = target_maxlen\n        self.num_classes = num_classes\n\n        self.enc_input = LandmarkEmbedding(num_hid=num_hid, maxlen=source_maxlen)\n        self.dec_input = TokenEmbedding(\n            num_vocab=num_classes, maxlen=target_maxlen, num_hid=num_hid\n        )\n\n        self.encoder = keras.Sequential(\n            [self.enc_input]\n            + [\n                TransformerEncoder(num_hid, num_head, num_feed_forward)\n                for _ in range(num_layers_enc)\n            ]\n        )\n\n        for i in range(num_layers_dec):\n            setattr(\n                self,\n                f\"dec_layer_{i}\",\n                TransformerDecoder(num_hid, num_head, num_feed_forward),\n            )\n\n        self.classifier = layers.Dense(num_classes)  \n        \n        \n\n        self.loss_metric = keras.metrics.Mean(name=\"loss\")\n        \n        # ISSO AQUI E NO TRAINER\n        self.acc_metric = keras.metrics.Mean(name=\"edit_dist\")\n\n\n    def decode(self, enc_out, target, training):\n        y = self.dec_input(target)\n        for i in range(self.num_layers_dec):\n            y = getattr(self, f\"dec_layer_{i}\")(enc_out, y, training)\n        return y\n\n    def call(self, inputs, training):\n        source = inputs[0]\n        target = inputs[1]\n        x = self.encoder(source, training)\n        y = self.decode(x, target, training)\n        return self.classifier(y)\n\n    @property\n    def metrics(self):\n        return [self.loss_metric]\n\n\n\n    def train_step(self, batch):\n        \"\"\"Processes one batch inside model.fit().\"\"\"\n        source = batch[0]\n        target = batch[1]\n\n        input_shape = tf.shape(target)\n        batch_size = input_shape[0]\n        \n        dec_input = target[:, :-1]\n        dec_target = target[:, 1:]\n        \n        with tf.GradientTape() as tape:\n            preds = self([source, dec_input])\n            one_hot = tf.one_hot(dec_target, depth=self.num_classes)\n            mask = tf.math.logical_not(tf.math.equal(dec_target, pad_token_idx))\n            loss = self.compiled_loss(one_hot, preds, sample_weight=mask)\n            \n        # Implicitamente está no Learner\n        \n        trainable_vars = self.trainable_variables\n        gradients = tape.gradient(loss, trainable_vars)\n        self.optimizer.apply_gradients(zip(gradients, trainable_vars))\n        \n        # Computes the Levenshtein distance between sequences since the evaluation\n        # metric for this contest is the normalized total levenshtein distance.\n        edit_dist = tf.edit_distance(tf.sparse.from_dense(target), \n                                     tf.sparse.from_dense(tf.cast(tf.argmax(preds, axis=1), tf.int32)))\n        edit_dist = tf.reduce_mean(edit_dist)\n        \n        self.acc_metric.update_state(edit_dist)\n        self.loss_metric.update_state(loss)\n        \n        return {\"loss\": self.loss_metric.result(), \"edit_dist\": self.acc_metric.result()}\n\n    def test_step(self, batch):        \n        source = batch[0]\n        target = batch[1]\n\n        input_shape = tf.shape(target)\n        batch_size = input_shape[0]\n        \n        dec_input = target[:, :-1]\n        dec_target = target[:, 1:]\n        \n        preds = self([source, dec_input])\n        one_hot = tf.one_hot(dec_target, depth=self.num_classes)\n        mask = tf.math.logical_not(tf.math.equal(dec_target, pad_token_idx))\n        loss = self.compiled_loss(one_hot, preds, sample_weight=mask)\n        \n        # Computes the Levenshtein distance between sequences since the evaluation\n        # metric for this contest is the normalized total levenshtein distance.\n        edit_dist = tf.edit_distance(tf.sparse.from_dense(target), \n                                     tf.sparse.from_dense(tf.cast(tf.argmax(preds, axis=1), tf.int32)))\n        edit_dist = tf.reduce_mean(edit_dist)\n        self.acc_metric.update_state(edit_dist)\n        self.loss_metric.update_state(loss)\n        return {\"loss\": self.loss_metric.result(), \"edit_dist\": self.acc_metric.result()}\n","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:01.77868Z","iopub.execute_input":"2023-08-24T05:03:01.779269Z","iopub.status.idle":"2023-08-24T05:03:01.803902Z","shell.execute_reply.started":"2023-08-24T05:03:01.779195Z","shell.execute_reply":"2023-08-24T05:03:01.802876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_model = Transformer()","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:01.805608Z","iopub.execute_input":"2023-08-24T05:03:01.806698Z","iopub.status.idle":"2023-08-24T05:03:01.880628Z","shell.execute_reply.started":"2023-08-24T05:03:01.806664Z","shell.execute_reply":"2023-08-24T05:03:01.879721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch = next(iter(valid_ds)) \nmy_model(batch)\n\nmy_model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:01.882353Z","iopub.execute_input":"2023-08-24T05:03:01.882945Z","iopub.status.idle":"2023-08-24T05:03:02.587925Z","shell.execute_reply.started":"2023-08-24T05:03:01.882909Z","shell.execute_reply":"2023-08-24T05:03:02.586775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Learner","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nclass Learner:\n    def __init__(self):\n        self.model = Transformer()\n        self.optimizer = keras.optimizers.Adam(0.0001)\n\n    def generate(self, source, target_start_token_idx):\n        \"\"\"Performs inference over one batch of inputs using greedy decoding.\"\"\"\n        bs = tf.shape(source)[0]\n        enc = self.model.encoder(source, training = False)\n        dec_input = tf.ones((bs, 1), dtype=tf.int32) * target_start_token_idx\n        dec_logits = []\n        for i in range(self.model.target_maxlen - 1):\n            dec_out = self.model.decode(enc, dec_input, training = False)\n            logits = self.model.classifier(dec_out)\n            logits = tf.argmax(logits, axis=-1, output_type=tf.int32)\n            last_logit = logits[:, -1][..., tf.newaxis]\n            dec_logits.append(last_logit)\n            dec_input = tf.concat([dec_input, last_logit], axis=-1)\n        return dec_input\n    \n#     def predict(self, batch):\n#         source = batch[0]\n#         target = batch[1]\n\n#         input_shape = tf.shape(target)\n#         batch_size = input_shape[0]\n        \n#         dec_input = target[:, :-1]\n#         dec_target = target[:, 1:]\n        \n#         preds = self.model.call([source, dec_input])\n        \n#         return preds\n\n","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:02.589934Z","iopub.execute_input":"2023-08-24T05:03:02.59076Z","iopub.status.idle":"2023-08-24T05:03:02.60198Z","shell.execute_reply.started":"2023-08-24T05:03:02.590718Z","shell.execute_reply":"2023-08-24T05:03:02.601139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_learner = Learner()","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:02.603656Z","iopub.execute_input":"2023-08-24T05:03:02.604431Z","iopub.status.idle":"2023-08-24T05:03:02.668938Z","shell.execute_reply.started":"2023-08-24T05:03:02.604375Z","shell.execute_reply":"2023-08-24T05:03:02.667972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"https://keras.io/api/optimizers/adam/","metadata":{}},{"cell_type":"code","source":"print(\"Optimizer Configuration:\")\nprint(f\"Learning Rate: {my_learner.optimizer.learning_rate.numpy()}\")\nprint(f\"Beta 1: {my_learner.optimizer.beta_1}\")\nprint(f\"Beta 2: {my_learner.optimizer.beta_2}\")\nprint(f\"Epsilon: {my_learner.optimizer.epsilon}\")\nprint(f\"Weight_decay: {my_learner.optimizer.weight_decay}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:02.670301Z","iopub.execute_input":"2023-08-24T05:03:02.670726Z","iopub.status.idle":"2023-08-24T05:03:02.678959Z","shell.execute_reply.started":"2023-08-24T05:03:02.670691Z","shell.execute_reply":"2023-08-24T05:03:02.677898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluator","metadata":{}},{"cell_type":"code","source":"class Evaluator:\n    def __init__(self):\n        self.loss_fn = tf.keras.losses.CategoricalCrossentropy(from_logits=True, label_smoothing=0.1)\n\n#     def get_loss(self, y, y_hat):\n#         return self.loss_fn(y_hat, y)","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:02.680809Z","iopub.execute_input":"2023-08-24T05:03:02.681317Z","iopub.status.idle":"2023-08-24T05:03:02.687508Z","shell.execute_reply.started":"2023-08-24T05:03:02.681251Z","shell.execute_reply":"2023-08-24T05:03:02.686357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_evaluator = Evaluator()","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:02.689429Z","iopub.execute_input":"2023-08-24T05:03:02.689995Z","iopub.status.idle":"2023-08-24T05:03:02.696829Z","shell.execute_reply.started":"2023-08-24T05:03:02.68996Z","shell.execute_reply":"2023-08-24T05:03:02.695744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"https://keras.io/api/losses/probabilistic_losses/#categoricalcrossentropy-class","metadata":{}},{"cell_type":"code","source":"print(\"Loss Function Configuration:\")\nprint(f\"name: {my_evaluator.loss_fn.name}\")\nprint(f\"Reduction Type: {my_evaluator.loss_fn.reduction}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:02.698452Z","iopub.execute_input":"2023-08-24T05:03:02.698948Z","iopub.status.idle":"2023-08-24T05:03:02.706975Z","shell.execute_reply.started":"2023-08-24T05:03:02.69879Z","shell.execute_reply":"2023-08-24T05:03:02.705892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Trainer","metadata":{}},{"cell_type":"code","source":"class DisplayOutputs(keras.callbacks.Callback):\n    def __init__(self, batch,learner: Learner, idx_to_token, target_start_token_idx=60, target_end_token_idx=61):\n        \"\"\"Displays a batch of outputs after every 4 epoch\n\n        Args:\n            batch: A test batch\n            idx_to_token: A List containing the vocabulary tokens corresponding to their indices\n            target_start_token_idx: A start token index in the target vocabulary\n            target_end_token_idx: An end token index in the target vocabulary\n        \"\"\"\n        self.batch = batch\n        self.learner = learner\n        self.target_start_token_idx = target_start_token_idx\n        self.target_end_token_idx = target_end_token_idx\n        self.idx_to_char = idx_to_token\n\n    def on_epoch_end(self, epoch, logs=None):\n        if epoch % 4 != 0:\n            return\n        source = self.batch[0]\n        target = self.batch[1].numpy()\n        bs = tf.shape(source)[0]\n        preds = self.learner.generate(source, self.target_start_token_idx)\n        preds = preds.numpy()\n        for i in range(bs):\n            target_text = \"\".join([self.idx_to_char[_] for _ in target[i, :]])\n            prediction = \"\"\n            for idx in preds[i, :]:\n                prediction += self.idx_to_char[idx]\n                if idx == self.target_end_token_idx:\n                    break\n            print(f\"\\ntarget:     {target_text.replace('-','')}\")\n            print(f\"prediction: {prediction}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:02.708867Z","iopub.execute_input":"2023-08-24T05:03:02.709395Z","iopub.status.idle":"2023-08-24T05:03:02.721496Z","shell.execute_reply.started":"2023-08-24T05:03:02.709361Z","shell.execute_reply":"2023-08-24T05:03:02.720848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Trainer:\n    def __init__(self, data: Data, learner: Learner, evaluator: Evaluator):\n        self.data = data\n        self.learner = learner\n        self.evaluator = evaluator\n  \n        # The vocabulary to convert predicted indices into characters\n        idx_to_char = list(char_to_num.keys())\n        batch = next(iter(valid_ds))\n        # set the arguments as per vocabulary index for '<' and '>'\n        self.display_cb = DisplayOutputs(batch,self.learner, idx_to_char, target_start_token_idx=char_to_num['<'], target_end_token_idx=char_to_num['>']) \n\n        \n    def run(self, num_epochs:int):\n        self.learner.model.compile(optimizer=self.learner.optimizer, loss=self.evaluator.loss_fn)\n        \n        valid_ds = self.data.get_dataloader(train=False)\n        train_ds = self.data.get_dataloader()\n        \n        history = self.learner.model.fit(train_ds, validation_data=valid_ds, callbacks=[self.display_cb], epochs=num_epochs)\n        \n        return history","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:02.722947Z","iopub.execute_input":"2023-08-24T05:03:02.724109Z","iopub.status.idle":"2023-08-24T05:03:02.734339Z","shell.execute_reply.started":"2023-08-24T05:03:02.724075Z","shell.execute_reply":"2023-08-24T05:03:02.733532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = Trainer(my_data, my_learner, my_evaluator)","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:02.735545Z","iopub.execute_input":"2023-08-24T05:03:02.738076Z","iopub.status.idle":"2023-08-24T05:03:03.026566Z","shell.execute_reply.started":"2023-08-24T05:03:02.738033Z","shell.execute_reply":"2023-08-24T05:03:03.025509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = train.run(num_epochs = 30)","metadata":{"execution":{"iopub.status.busy":"2023-08-24T05:03:03.032072Z","iopub.execute_input":"2023-08-24T05:03:03.03247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.legend(['training loss', 'val_loss'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}