{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Notebooks I'm using as references:** \n\n[LB 0.67] one pytorch transformer solution<br>\nZ by HP Data Science Global Ambassador<br>\nhttps://www.kaggle.com/code/hengck23/lb-0-67-one-pytorch-transformer-solution\n\n\nASLFR EDA + Preprocessing Dataset<br>\nMARK WIJKHUIZEN<br>\nhttps://www.kaggle.com/code/markwijkhuizen/aslfr-eda-preprocessing-dataset\n\n**Discussions I'm using as references:**\n\nCV Leaderboard<br>\nMARK WIJKHUIZEN<br>\nhttps://www.kaggle.com/competitions/asl-fingerspelling/discussion/411060","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport seaborn as sn\nimport tensorflow as tf\n\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import train_test_split, GroupShuffleSplit \n\nimport glob\nimport sys\nimport os\nimport math\nimport gc\nimport sys\nimport sklearn\nimport time\nimport json\n\n# TQDM Progress Bar With Pandas Apply Function\ntqdm.pandas()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-21T18:24:56.465633Z","iopub.execute_input":"2023-05-21T18:24:56.466469Z","iopub.status.idle":"2023-05-21T18:25:05.883904Z","shell.execute_reply.started":"2023-05-21T18:24:56.466409Z","shell.execute_reply":"2023-05-21T18:25:05.883017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# If Notebook Is Run By Committing or In Interactive Mode For Development\nIS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\n# Describe Statistics Percentiles\nPERCENTILES = [0.01, 0.05, 0.25, 0.50, 0.75, 0.95, 0.99, 0.999]\n# Global Random Seed\nSEED = 42\n# Number of Frames to resize recording to\nN_TARGET_FRAMES = 256\n# Global debug flag, takes subset of train\nDEBUG = False","metadata":{"execution":{"iopub.status.busy":"2023-05-21T18:25:05.885735Z","iopub.execute_input":"2023-05-21T18:25:05.886915Z","iopub.status.idle":"2023-05-21T18:25:05.892586Z","shell.execute_reply.started":"2023-05-21T18:25:05.88688Z","shell.execute_reply":"2023-05-21T18:25:05.891514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv').head(5000)","metadata":{"execution":{"iopub.status.busy":"2023-05-21T18:25:05.899159Z","iopub.execute_input":"2023-05-21T18:25:05.899541Z","iopub.status.idle":"2023-05-21T18:25:06.067837Z","shell.execute_reply.started":"2023-05-21T18:25:05.899499Z","shell.execute_reply":"2023-05-21T18:25:06.066858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get complete file path to file\ndef get_file_path(path):\n    return f'/kaggle/input/asl-fingerspelling/{path}'\n\ntrain['file_path'] = train['path'].apply(get_file_path)","metadata":{"execution":{"iopub.status.busy":"2023-05-21T18:25:06.069088Z","iopub.execute_input":"2023-05-21T18:25:06.070055Z","iopub.status.idle":"2023-05-21T18:25:06.082495Z","shell.execute_reply.started":"2023-05-21T18:25:06.070016Z","shell.execute_reply":"2023-05-21T18:25:06.081264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_idxs(df, words_pos, words_neg=[], ret_names=True):\n    idxs = []\n    names = []\n    for col_idx, col in enumerate(df.columns):\n        # Check if column name contains all words\n        if all([w in col for w in words_pos]) and all([w not in col for w in words_neg]):\n            idxs.append(col_idx)\n            names.append(col)\n    # Convert to Numpy arrays\n    idxs = np.array(idxs)\n    names = np.array(names)\n    # Returns either both column indices and names\n    if ret_names:\n        return idxs, names\n    # Or only columns indices\n    else:\n        return idxs","metadata":{"execution":{"iopub.status.busy":"2023-05-21T18:25:06.083781Z","iopub.execute_input":"2023-05-21T18:25:06.084118Z","iopub.status.idle":"2023-05-21T18:25:06.094919Z","shell.execute_reply.started":"2023-05-21T18:25:06.084082Z","shell.execute_reply":"2023-05-21T18:25:06.093868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read First Parquet File\nexample_parquet_df = pd.read_parquet(train['file_path'][0])\n\n# Each parquet file contains 1000 recordings\nprint(f'# Unique Recording: {example_parquet_df.index.nunique()}')\n# Display DataFrame layout\ndisplay(example_parquet_df.head())","metadata":{"execution":{"iopub.status.busy":"2023-05-21T18:25:06.096098Z","iopub.execute_input":"2023-05-21T18:25:06.096434Z","iopub.status.idle":"2023-05-21T18:25:22.873856Z","shell.execute_reply.started":"2023-05-21T18:25:06.096406Z","shell.execute_reply":"2023-05-21T18:25:22.872755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Landmark Indices for Left/Right hand without z axis in raw data\nLEFT_HAND_IDXS0, LEFT_HAND_NAMES0 = get_idxs(example_parquet_df, ['left_hand'], ['z'])\nRIGHT_HAND_IDXS0, RIGHT_HAND_NAMES0 = get_idxs(example_parquet_df, ['right_hand'], ['z'])\nCOLUMNS = np.concatenate((LEFT_HAND_NAMES0, RIGHT_HAND_NAMES0))\nN_COLS0 = len(COLUMNS)\n# Only X/Y axes are used\nN_DIMS0 = 2\n\nprint(f'N_COLS0: {N_COLS0}')","metadata":{"execution":{"iopub.status.busy":"2023-05-21T18:25:22.875567Z","iopub.execute_input":"2023-05-21T18:25:22.876184Z","iopub.status.idle":"2023-05-21T18:25:22.884195Z","shell.execute_reply.started":"2023-05-21T18:25:22.876144Z","shell.execute_reply":"2023-05-21T18:25:22.883424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_SAMPLES = len(train)\nN_COLS0 = len(COLUMNS)","metadata":{"execution":{"iopub.status.busy":"2023-05-21T18:25:22.885891Z","iopub.execute_input":"2023-05-21T18:25:22.88654Z","iopub.status.idle":"2023-05-21T18:25:22.907237Z","shell.execute_reply.started":"2023-05-21T18:25:22.886511Z","shell.execute_reply":"2023-05-21T18:25:22.906391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n    Tensorflow layer to process data in TFLite\n    Data needs to be processed in the model itself, so we can not use Python\n\"\"\" \nclass PreprocessLayer(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayer, self).__init__()\n        self.normalisation_correction = tf.constant(\n                    # Add 0.50 to x coordinates of left hand (original right hand) and substract 0.50 of right hand (original left hand)\n                     [0.50 if 'x' in name else 0.00 for name in LEFT_HAND_NAMES0],\n                dtype=tf.float32,\n            )\n    \n    @tf.function(\n        input_signature=(tf.TensorSpec(shape=[None,N_COLS0], dtype=tf.float32),),\n    )\n    def call(self, data0):\n        # Number of Frames in Video\n        N_FRAMES0 = tf.shape(data0)[0]\n        \n        # Find dominant hand\n        left_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS, axis=1)), 0, 1))\n        right_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS, axis=1)), 0, 1))\n        left_dominant = left_hand_sum >= right_hand_sum\n        \n        # Count non NaN Hand values in each frame\n        if left_dominant:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                    tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS, axis=1)), 0, 1),\n                    axis=[1],\n                )\n        else:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                    tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS, axis=1)), 0, 1),\n                    axis=[1],\n                )\n        # Frames With Coordinates for hand\n        non_empty_frames_idxs = tf.where(frames_hands_non_nan_sum > 0)\n        non_empty_frames_idxs = tf.squeeze(non_empty_frames_idxs, axis=1)\n        # Filter data on frames with coordinates for hand\n        data = tf.gather(data0, non_empty_frames_idxs, axis=0)\n        \n        # Cast Indices in float32 to be compatible with Tensorflow Lite\n        non_empty_frames_idxs = tf.cast(non_empty_frames_idxs, tf.float32)\n        # Normalize to start with 0\n        non_empty_frames_idxs -= tf.reduce_min(non_empty_frames_idxs)\n        \n        # Number of Frames in Filtered Video\n        N_FRAMES = tf.shape(data)[0]\n        \n        # Gather Relevant Landmark Columns\n        if left_dominant:\n            data = tf.gather(data, LEFT_HAND_IDXS, axis=1)\n        else:\n            data = tf.gather(data, RIGHT_HAND_IDXS, axis=1)\n            data = (\n                    self.normalisation_correction + (\n                        (data - self.normalisation_correction) * tf.where(self.normalisation_correction != 0, -1.0, 1.0))\n                )\n            \n        # Fill NaN Values With 0\n        data = tf.where(tf.math.is_nan(data), 0.0, data)\n        # Resize Video\n        data = tf.image.resize(\n            data[:,:,tf.newaxis],\n            [N_TARGET_FRAMES, N_COLS],\n            method=tf.image.ResizeMethod.BILINEAR,\n            antialias=False,\n        )\n        data = tf.squeeze(data, axis=[2])\n        # Resize Non Empty Frame Indices\n        non_empty_frames_idxs = tf.image.resize(\n            non_empty_frames_idxs[:,tf.newaxis, tf.newaxis],\n            [N_TARGET_FRAMES, 1],\n            method=tf.image.ResizeMethod.BILINEAR,\n            antialias=False,\n        )\n        non_empty_frames_idxs = tf.squeeze(non_empty_frames_idxs, axis=[1,2])\n        \n        return data, non_empty_frames_idxs\n\n    \npreprocess_layer = PreprocessLayer()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T18:25:22.910609Z","iopub.execute_input":"2023-05-21T18:25:22.911311Z","iopub.status.idle":"2023-05-21T18:25:23.017042Z","shell.execute_reply.started":"2023-05-21T18:25:22.911278Z","shell.execute_reply":"2023-05-21T18:25:23.016202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Landmark Indices in subset of dataframe with only COLUMNS selected\nLEFT_HAND_IDXS = np.argwhere(np.isin(COLUMNS, LEFT_HAND_NAMES0)).squeeze()\nRIGHT_HAND_IDXS = np.argwhere(np.isin(COLUMNS, RIGHT_HAND_NAMES0)).squeeze()\nN_COLS = LEFT_HAND_IDXS.size\n# Only X/Y axes are used\nN_DIMS = 2\n\nprint(f'N_COLS: {N_COLS}')","metadata":{"execution":{"iopub.status.busy":"2023-05-21T18:25:23.018167Z","iopub.execute_input":"2023-05-21T18:25:23.018975Z","iopub.status.idle":"2023-05-21T18:25:23.027367Z","shell.execute_reply.started":"2023-05-21T18:25:23.018944Z","shell.execute_reply":"2023-05-21T18:25:23.026274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split Phrase To Char Tuple\ntrain['phrase_char'] = train['phrase'].apply(tuple)\n# Character Length of Phrase\ntrain['phrase_char_len'] = train['phrase_char'].apply(len)\n\n# Maximum Input Length\nMAX_PHRASE_LENGTH = train['phrase_char_len'].max()\nprint(f'MAX_PHRASE_LENGTH: {MAX_PHRASE_LENGTH}')","metadata":{"execution":{"iopub.status.busy":"2023-05-21T18:25:23.028844Z","iopub.execute_input":"2023-05-21T18:25:23.029376Z","iopub.status.idle":"2023-05-21T18:25:23.051033Z","shell.execute_reply.started":"2023-05-21T18:25:23.029346Z","shell.execute_reply":"2023-05-21T18:25:23.049412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use Set to keep track of unique characters in phrases\nUNIQUE_CHARACTERS = set()\n\nfor phrase in tqdm(train['phrase_char']):\n    for c in phrase:\n        UNIQUE_CHARACTERS.add(c)\n        \n# Sorted Unique Character\nUNIQUE_CHARACTERS = np.array(sorted(UNIQUE_CHARACTERS))\n# Number of Unique Characters\nN_UNIQUE_CHARACTERS = len(UNIQUE_CHARACTERS)\nprint(f'N_UNIQUE_CHARACTERS: {N_UNIQUE_CHARACTERS}')","metadata":{"execution":{"iopub.status.busy":"2023-05-21T18:25:23.052544Z","iopub.execute_input":"2023-05-21T18:25:23.052937Z","iopub.status.idle":"2023-05-21T18:25:23.096932Z","shell.execute_reply.started":"2023-05-21T18:25:23.052906Z","shell.execute_reply":"2023-05-21T18:25:23.096174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Target Arrays Processed Input Videos\nX = np.zeros([N_SAMPLES, N_TARGET_FRAMES, N_COLS], dtype=np.float32)\n# Frame Indices\nNON_EMPTY_FRAME_IDXS = np.zeros([N_SAMPLES, N_TARGET_FRAMES], dtype=np.uint16)\n# Ordinally Encoded Target With value 59 for pad token\ny = np.full(shape=[N_SAMPLES, MAX_PHRASE_LENGTH], fill_value=N_UNIQUE_CHARACTERS, dtype=np.int8)\n\n# Train DataFrame indexed by sequence_id to convenientlyy lookup recording data\ntrain_squence_id = train.set_index('sequence_id')","metadata":{"execution":{"iopub.status.busy":"2023-05-21T18:25:23.098161Z","iopub.execute_input":"2023-05-21T18:25:23.098693Z","iopub.status.idle":"2023-05-21T18:25:23.106926Z","shell.execute_reply.started":"2023-05-21T18:25:23.098662Z","shell.execute_reply":"2023-05-21T18:25:23.105806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read Character to Ordinal Encoding Mapping\nwith open('/kaggle/input/asl-fingerspelling/character_to_prediction_index.json') as json_file:\n    CHAR2ORD = json.load(json_file)\n    \n# Character to Ordinal Encoding Mapping   \ndisplay(pd.Series(CHAR2ORD).to_frame('Ordinal Encoding'))","metadata":{"execution":{"iopub.status.busy":"2023-05-21T18:25:23.108343Z","iopub.execute_input":"2023-05-21T18:25:23.108681Z","iopub.status.idle":"2023-05-21T18:25:23.128965Z","shell.execute_reply.started":"2023-05-21T18:25:23.108653Z","shell.execute_reply":"2023-05-21T18:25:23.127977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of Unique Characters\nN_UNIQUE_CHARACTERS = len(CHAR2ORD)\nprint(f'N_UNIQUE_CHARACTERS: {N_UNIQUE_CHARACTERS}')","metadata":{"execution":{"iopub.status.busy":"2023-05-21T18:25:23.130256Z","iopub.execute_input":"2023-05-21T18:25:23.130664Z","iopub.status.idle":"2023-05-21T18:25:23.135511Z","shell.execute_reply.started":"2023-05-21T18:25:23.130633Z","shell.execute_reply":"2023-05-21T18:25:23.134603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# All Unique Parquet Files\nUNIQUE_FILE_PATHS = pd.Series(train['file_path'].unique())\n# Counter to keep track of sample\nrow = 0\n\n# Fill Arrays\nfor idx, file_path in enumerate(tqdm(UNIQUE_FILE_PATHS)):\n    df = pd.read_parquet(file_path)\n    for group, group_df in df.groupby('sequence_id'):\n        # Get Processed Frames and non empty frame indices\n        data, non_empty_frames_idxs = preprocess_layer(group_df[COLUMNS].values)\n        X[row] = data\n        NON_EMPTY_FRAME_IDXS[row] = non_empty_frames_idxs\n        # Add Target By Ordinally Encoding Characters\n        phrase_char = train_squence_id.loc[group, 'phrase_char']\n        for col, char in enumerate(phrase_char):\n            y[row, col] = CHAR2ORD.get(char)\n            \n        row += 1","metadata":{"execution":{"iopub.status.busy":"2023-05-21T18:25:23.136865Z","iopub.execute_input":"2023-05-21T18:25:23.137615Z","iopub.status.idle":"2023-05-21T18:26:42.213192Z","shell.execute_reply.started":"2023-05-21T18:25:23.137579Z","shell.execute_reply":"2023-05-21T18:26:42.212362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example target, note the phrase is padded with the pad token 59\nprint(f'Example Target: {y[0]}')","metadata":{"execution":{"iopub.status.busy":"2023-05-21T18:26:42.21451Z","iopub.execute_input":"2023-05-21T18:26:42.215031Z","iopub.status.idle":"2023-05-21T18:26:42.220057Z","shell.execute_reply.started":"2023-05-21T18:26:42.214998Z","shell.execute_reply":"2023-05-21T18:26:42.219005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save X/y\nnp.save('X.npy', X)\nnp.save('y.npy', y)\nnp.save('NON_EMPTY_FRAME_IDXS.npy', NON_EMPTY_FRAME_IDXS)\n# Save Validation\nsplitter = GroupShuffleSplit(test_size=0.10, n_splits=2, random_state=SEED)\nPARTICIPANT_IDS = train['participant_id'].values\ntrain_idxs, val_idxs = next(splitter.split(X, y, groups=PARTICIPANT_IDS))\n\n# Save Train\nnp.save('X_train.npy', X[train_idxs])\nnp.save('y_train.npy', y[train_idxs])\nnp.save('NON_EMPTY_FRAME_IDXS_TRAIN.npy', NON_EMPTY_FRAME_IDXS[train_idxs])\n# Save Validation\nnp.save('X_val.npy', X[val_idxs])\nnp.save('y_val.npy', y[val_idxs])\nnp.save('NON_EMPTY_FRAME_IDXS_VAL.npy', NON_EMPTY_FRAME_IDXS[val_idxs])\n# Verify Train/Val is correctly split by participan id\nprint(f'Patient ID Intersection Train/Val: {set(PARTICIPANT_IDS[train_idxs]).intersection(PARTICIPANT_IDS[val_idxs])}')\n# Train/Val Sizes\nprint(f'# Train Samples: {len(train_idxs)}, # Val Samples: {len(val_idxs)}')","metadata":{"execution":{"iopub.status.busy":"2023-05-21T18:27:46.536818Z","iopub.execute_input":"2023-05-21T18:27:46.537241Z","iopub.status.idle":"2023-05-21T18:27:46.990838Z","shell.execute_reply.started":"2023-05-21T18:27:46.537206Z","shell.execute_reply":"2023-05-21T18:27:46.98968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = X[train_idxs], X[val_idxs], y[train_idxs], y[val_idxs]","metadata":{"execution":{"iopub.status.busy":"2023-05-21T19:10:19.473152Z","iopub.execute_input":"2023-05-21T19:10:19.473577Z","iopub.status.idle":"2023-05-21T19:10:19.740299Z","shell.execute_reply.started":"2023-05-21T19:10:19.473541Z","shell.execute_reply":"2023-05-21T19:10:19.739275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So far, I've only adjusted the code developed by MARK WIJKHUIZEN, the link to MARK WIJKHUIZEN's work is found at the beginning of this notebook, now it seems to me that the work would be a prediction.","metadata":{}},{"cell_type":"markdown","source":"**Honestly, I don't have a clue on how to proceed... If anyone can help me out, I'd appreciate it.**","metadata":{}}]}