{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Predicting the output length\nIn this notebook I designed a model to solve the subproblem of predcitng the number of output length. \nI use a series of 1D CNN to achieve ~0.9 accuracy.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        continue\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-15T20:37:30.487201Z","iopub.execute_input":"2023-06-15T20:37:30.487599Z","iopub.status.idle":"2023-06-15T20:37:30.513035Z","shell.execute_reply.started":"2023-06-15T20:37:30.48757Z","shell.execute_reply":"2023-06-15T20:37:30.512137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom os import path\nimport glob\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:37:31.807013Z","iopub.execute_input":"2023-06-15T20:37:31.807441Z","iopub.status.idle":"2023-06-15T20:37:31.812818Z","shell.execute_reply.started":"2023-06-15T20:37:31.807407Z","shell.execute_reply":"2023-06-15T20:37:31.811706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_hand_lmrks = 21\nhandedness = ['left', 'right']\nselected_columns = []\nfor hand in handedness:\n    for i in range(n_hand_lmrks):\n        selected_columns.append(f'x_{hand}_hand_{i}')\n        selected_columns.append(f'y_{hand}_hand_{i}')\n","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:37:33.185498Z","iopub.execute_input":"2023-06-15T20:37:33.18587Z","iopub.status.idle":"2023-06-15T20:37:33.191248Z","shell.execute_reply.started":"2023-06-15T20:37:33.185841Z","shell.execute_reply":"2023-06-15T20:37:33.190308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_path = '/kaggle/input/asl-fingerspelling/train_landmarks'\nparquet_files = glob.glob(path.join(data_path, '*.parquet'))\nvalid_files = [parquet_files[0]]\ntrain_files = parquet_files[1:4]","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:37:33.615915Z","iopub.execute_input":"2023-06-15T20:37:33.616597Z","iopub.status.idle":"2023-06-15T20:37:33.623365Z","shell.execute_reply.started":"2023-06-15T20:37:33.616559Z","shell.execute_reply":"2023-06-15T20:37:33.622405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_path = '/kaggle/input/asl-fingerspelling/train.csv'\nlabel_df = pd.read_csv(label_path)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:37:34.009724Z","iopub.execute_input":"2023-06-15T20:37:34.010406Z","iopub.status.idle":"2023-06-15T20:37:34.092216Z","shell.execute_reply.started":"2023-06-15T20:37:34.010366Z","shell.execute_reply":"2023-06-15T20:37:34.091261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"channels = len(selected_columns)\nfilters = 128\nkernel_size = 10\noutput_len = 32\n\nexp = 4\ndropout_rate = 0.1\ninput_len = output_len * (2 ** exp)\nbatch_size = 128\nepochs = 5\nruns_per_epoch = 5\nsteps =  1\n","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:37:34.293227Z","iopub.execute_input":"2023-06-15T20:37:34.29479Z","iopub.status.idle":"2023-06-15T20:37:34.300525Z","shell.execute_reply.started":"2023-06-15T20:37:34.294753Z","shell.execute_reply":"2023-06-15T20:37:34.299386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_steps = 0\nfor i in range(0, len(train_files), steps):\n    num_samples = 0\n    for f in train_files[i:i+steps]:\n        df = pd.read_parquet(f)\n        num_samples += len(df.index.unique())\n    num_steps += num_samples // batch_size\n    if num_steps % batch_size != 0:\n        num_steps += 1\n        \nvalid_num_steps = 0 \nfor i in range(0, len(valid_files), steps):\n    num_samples = 0\n    for f in valid_files[i:i+steps]:\n        df = pd.read_parquet(f)\n        num_samples += len(df.index.unique())\n    valid_num_steps += num_samples // batch_size\n    if num_steps % batch_size != 0:\n        valid_num_steps += 1\n","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:37:34.573743Z","iopub.execute_input":"2023-06-15T20:37:34.574107Z","iopub.status.idle":"2023-06-15T20:38:32.827185Z","shell.execute_reply.started":"2023-06-15T20:37:34.574078Z","shell.execute_reply":"2023-06-15T20:38:32.826205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class conv1DBlockLn(tf.keras.layers.Layer):\n    def __init__(self, filters, kernel_size, strides, padding, activation, use_bias, name, \n                 dropout_rate=0.2):\n        super(conv1DBlockLn, self).__init__(name=name)\n        self.conv1d = tf.keras.layers.Conv1D(filters=filters, kernel_size=kernel_size, \n                                             strides=strides, padding=padding, \n                                             activation=activation, use_bias=use_bias, \n                                             name=name)\n        self.ln = tf.keras.layers.LayerNormalization(name=name)\n        self.dropout = tf.keras.layers.Dropout(rate=dropout_rate, name=name)\n        self.pool = tf.keras.layers.MaxPooling1D(pool_size=2, strides=2, padding='valid', \n                                         name=name)\n    def call(self, inputs, training=False):\n        x = self.conv1d(inputs)\n        x = self.ln(x, training=training)\n        x = self.pool(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:38:32.830894Z","iopub.execute_input":"2023-06-15T20:38:32.831298Z","iopub.status.idle":"2023-06-15T20:38:32.843215Z","shell.execute_reply.started":"2023-06-15T20:38:32.831263Z","shell.execute_reply":"2023-06-15T20:38:32.8423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_mask(output_len, channels, exp, filters, kernel_size, dropout_rate=0.2):\n    input_len = output_len * 2 ** exp\n    inputs = tf.keras.Input(shape=(input_len, channels), name='input')\n    x = inputs\n    for i in range(exp):\n        x = conv1DBlockLn(filters=filters, kernel_size=kernel_size, strides=1, padding='same',\n                        activation='relu', use_bias=False, \n                        dropout_rate = dropout_rate, name=f'conv1d_{i}')(x)\n    # x = EncoderLayer(d_model=filters, num_heads=8, dff=128*2, dropout_rate=dropout_rate)(x)\n    x = tf.keras.layers.Dense(units=filters,  activation='relu')(x)\n    x = tf.keras.layers.Dropout(rate=0.1 )(x)\n    x = tf.keras.layers.Dense(units=filters,  activation='relu')(x)\n    x = tf.keras.layers.Dropout(rate=0.8)(x)\n    x = tf.keras.layers.Dense(units=1, activation='sigmoid', name='sigmoid')(x)\n    x = tf.squeeze(x, axis=-1)\n\n    return tf.keras.Model(inputs=inputs, outputs=x, name='conv1d_model_mask')","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:38:32.844945Z","iopub.execute_input":"2023-06-15T20:38:32.845481Z","iopub.status.idle":"2023-06-15T20:38:32.858492Z","shell.execute_reply.started":"2023-06-15T20:38:32.845448Z","shell.execute_reply":"2023-06-15T20:38:32.857794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_param = {\n    'output_len': output_len,\n    'channels': channels,\n    'exp': exp,\n    'filters': filters,\n    'kernel_size': kernel_size,\n    'dropout_rate': dropout_rate,\n}\n","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:38:32.86112Z","iopub.execute_input":"2023-06-15T20:38:32.861731Z","iopub.status.idle":"2023-06-15T20:38:32.869637Z","shell.execute_reply.started":"2023-06-15T20:38:32.861698Z","shell.execute_reply":"2023-06-15T20:38:32.868748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = get_model_mask(**model_param)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:38:32.870948Z","iopub.execute_input":"2023-06-15T20:38:32.871576Z","iopub.status.idle":"2023-06-15T20:38:33.270869Z","shell.execute_reply.started":"2023-06-15T20:38:32.871545Z","shell.execute_reply":"2023-06-15T20:38:33.270147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomSchedule(tf.keras.optimizers.schedules.LearningRateSchedule):\n    def __init__(self, d_model, warmup_steps=4000):\n        super().__init__()\n\n        self.d_model = d_model\n        self.d_model = tf.cast(self.d_model, tf.float32)\n\n        self.warmup_steps = warmup_steps\n\n    def __call__(self, step):\n        step = tf.cast(step, dtype=tf.float32)\n        arg1 = tf.math.rsqrt(step)\n        arg2 = step * (self.warmup_steps ** -1.5)\n\n        return tf.math.rsqrt(self.d_model) * tf.math.minimum(arg1, arg2)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:38:33.271938Z","iopub.execute_input":"2023-06-15T20:38:33.272311Z","iopub.status.idle":"2023-06-15T20:38:33.287944Z","shell.execute_reply.started":"2023-06-15T20:38:33.272276Z","shell.execute_reply":"2023-06-15T20:38:33.286777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learning_rate = CustomSchedule(channels)\noptimizer = tf.keras.optimizers.Adam(learning_rate, beta_1=0.9, beta_2=0.98,\n                                    epsilon=1e-9)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:38:33.292267Z","iopub.execute_input":"2023-06-15T20:38:33.293757Z","iopub.status.idle":"2023-06-15T20:38:33.309769Z","shell.execute_reply.started":"2023-06-15T20:38:33.293725Z","shell.execute_reply":"2023-06-15T20:38:33.308605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef binary_loss(label, pred):\n    loss_object = tf.keras.losses.BinaryCrossentropy(\n        from_logits=True)\n    loss = loss_object(label, pred)\n\n    return loss\n\ndef binary_accuracy(label, pred):\n    pred = pred > 0.5\n    label = tf.cast(label, pred.dtype)\n    match = tf.cast(label == pred, tf.float32)\n    accuracy = tf.reduce_mean(match)\n    return accuracy","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:38:33.31114Z","iopub.execute_input":"2023-06-15T20:38:33.311728Z","iopub.status.idle":"2023-06-15T20:38:33.318355Z","shell.execute_reply.started":"2023-06-15T20:38:33.311698Z","shell.execute_reply":"2023-06-15T20:38:33.317352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer=optimizer, loss=binary_loss,\n              metrics=[binary_accuracy])","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:38:33.319799Z","iopub.execute_input":"2023-06-15T20:38:33.32043Z","iopub.status.idle":"2023-06-15T20:38:33.339692Z","shell.execute_reply.started":"2023-06-15T20:38:33.3204Z","shell.execute_reply":"2023-06-15T20:38:33.338645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Processing","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def adjust_seq_len(mat, seq_len):\n    if mat.shape[0] > seq_len:\n        mat = mat[:seq_len,:]\n    elif mat.shape[0] < seq_len:\n        mat = np.pad(mat, ((0, seq_len - mat.shape[0]), (0,0)), 'constant', constant_values = 0)\n    return mat\n\ndef get_features(parquet_paths, selected_columns, input_len):\n    df = pd.DataFrame()\n    for parquet_path in parquet_paths:\n        temp_df = pd.read_parquet(parquet_path, columns = selected_columns)\n        df = pd.concat([df, temp_df])   \n    df = df.fillna(0)\n    grouped_df = df.groupby(df.index).apply(lambda x: x.values)\n    grouped_values = grouped_df.apply(lambda x: adjust_seq_len(x, input_len))\n    features = np.stack(grouped_values.values).astype(np.float32)\n    return features, df.index.unique()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:38:33.343432Z","iopub.execute_input":"2023-06-15T20:38:33.343995Z","iopub.status.idle":"2023-06-15T20:38:33.353985Z","shell.execute_reply.started":"2023-06-15T20:38:33.343963Z","shell.execute_reply":"2023-06-15T20:38:33.352984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mask_data_generator(parquet_paths, label_path, input_length, output_length,\n                      selected_columns, batch_size, n_files_per_step= 1):\n    n_files = len(parquet_paths)\n    label_df = pd.read_csv(label_path)\n    label_df = label_df.set_index('sequence_id')\n    while True:\n        index = 0\n        while index < n_files:\n            batch_parquet_paths = parquet_paths[index:index + n_files_per_step]\n            index += n_files_per_step\n            features, sequence_ids = get_features(batch_parquet_paths, selected_columns, input_length)\n            # shuffle the features and sequence_ids\n            shuffle_idx = np.random.permutation(len(sequence_ids))\n            features = features[shuffle_idx]\n            sequence_ids = sequence_ids[shuffle_idx]\n            phrases = label_df.loc[sequence_ids]['phrase'].values\n            mask_seq = [np.ones(len(phrase)).astype(np.int32) for phrase in phrases]\n            labels = tf.keras.preprocessing.sequence.pad_sequences(mask_seq,\n                                                                value = 0,\n                                                                maxlen = output_length,\n                                                                padding = 'post',\n                                                                truncating = 'post')\n            for i in range(0, len(features), batch_size):\n                yield features[i:i+batch_size], labels[i:i+batch_size]","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:38:33.355504Z","iopub.execute_input":"2023-06-15T20:38:33.356098Z","iopub.status.idle":"2023-06-15T20:38:33.368721Z","shell.execute_reply.started":"2023-06-15T20:38:33.356067Z","shell.execute_reply":"2023-06-15T20:38:33.367722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_generator = mask_data_generator(train_files, label_path, input_len, \n                                     output_len, selected_columns, batch_size, \n                                     n_files_per_step= steps)\nvalid_data_generator = mask_data_generator(valid_files, label_path, input_len, \n                                     output_len, selected_columns, batch_size, \n                                     n_files_per_step= steps)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:38:33.371183Z","iopub.execute_input":"2023-06-15T20:38:33.371666Z","iopub.status.idle":"2023-06-15T20:38:33.38103Z","shell.execute_reply.started":"2023-06-15T20:38:33.371642Z","shell.execute_reply":"2023-06-15T20:38:33.380345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_data_generator, \n                    validation_data = valid_data_generator, \n                    epochs=epochs,\n                    steps_per_epoch=num_steps, \n                   validation_steps = valid_num_steps)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:38:33.382308Z","iopub.execute_input":"2023-06-15T20:38:33.382897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test model","metadata":{}},{"cell_type":"code","source":"def predict(model, inputs):\n    sigmoids = model(inputs, training=False)\n    binary = tf.cast(sigmoids > 0.5, tf.int32)\n    return binary.numpy()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = next(valid_data_generator)\ninputs = data[0]\nlabels = data[1]\nn_test_samples = 10\nreduced_inputs = inputs[:n_test_samples]\nreduced_labels = labels[:n_test_samples]\noutputs = predict(model, inputs)\nfor i in range(len(reduced_inputs)):\n    print(\"Prediction: \", outputs[i])\n    print(\"True label: \", reduced_labels[i])\n    print(\"\")\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-06-15T20:31:40.74513Z","iopub.execute_input":"2023-06-15T20:31:40.745816Z","iopub.status.idle":"2023-06-15T20:31:40.753658Z","shell.execute_reply.started":"2023-06-15T20:31:40.74577Z","shell.execute_reply":"2023-06-15T20:31:40.752591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}