{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Hello Fellow Kagglers,\n\nThis notebook gives an analysis of the competition dataset and demonstrates the preprocessing of the dataset, giving the X/y data to use for training.\n\nThe processing is as follows:\n\n1) Select dominant hand based on most number of non empty hand frames\n\n2) Filter out all frames with missing dominant hand coordinates\n\n3) Resize video to 256 frames\n\nThis is a work in progress and updates will follow.\n\nThe processed data could help with making a baseline.\n\nSoon, a training notebook will follow with the corresponding inference.\n\nExited to continue working on sign language!\n\n**Update V2**\n\n* Preprocessing pipeline that passes submission\n* Excluding samples with low frames per character ratio\n* Added phrase type\n\nTraining + Inference notebook coming soon.\n\n[Training + Inference Notebook](https://www.kaggle.com/markwijkhuizen/aslfr-transformer-training-inference)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport seaborn as sn\nimport tensorflow as tf\n\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import train_test_split, GroupShuffleSplit\nfrom pathlib import Path\n\nimport glob\nimport sys\nimport os\nimport math\nimport gc\nimport sys\nimport sklearn\nimport time\nimport json\nimport re\n\n# TQDM Progress Bar With Pandas Apply Function\ntqdm.pandas()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:45:27.199252Z","iopub.execute_input":"2023-06-15T11:45:27.199623Z","iopub.status.idle":"2023-06-15T11:45:36.448409Z","shell.execute_reply.started":"2023-06-15T11:45:27.199595Z","shell.execute_reply":"2023-06-15T11:45:36.447369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Character To Ordinal Encoding","metadata":{}},{"cell_type":"code","source":"# Read Character to Ordinal Encoding Mapping\nwith open('/kaggle/input/asl-fingerspelling/character_to_prediction_index.json') as json_file:\n    CHAR2ORD = json.load(json_file)\n    \n# Character to Ordinal Encoding Mapping   \ndisplay(pd.Series(CHAR2ORD).to_frame('Ordinal Encoding'))","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:45:36.450496Z","iopub.execute_input":"2023-06-15T11:45:36.45134Z","iopub.status.idle":"2023-06-15T11:45:36.494437Z","shell.execute_reply.started":"2023-06-15T11:45:36.451306Z","shell.execute_reply":"2023-06-15T11:45:36.493612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of Unique Characters\nN_UNIQUE_CHARACTERS = len(CHAR2ORD)\nprint(f'N_UNIQUE_CHARACTERS: {N_UNIQUE_CHARACTERS}')","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:45:36.495616Z","iopub.execute_input":"2023-06-15T11:45:36.496131Z","iopub.status.idle":"2023-06-15T11:45:36.501393Z","shell.execute_reply.started":"2023-06-15T11:45:36.496099Z","shell.execute_reply":"2023-06-15T11:45:36.500328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Global Config","metadata":{}},{"cell_type":"code","source":"# If Notebook Is Run By Committing or In Interactive Mode For Development\nIS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\n# Describe Statistics Percentiles\nPERCENTILES = [0.01, 0.10, 0.05, 0.25, 0.50, 0.75, 0.90, 0.95, 0.99, 0.999]\n# Global Random Seed\nSEED = 42\n# Number of Frames to resize recording to\nN_TARGET_FRAMES = 128\n# Global debug flag, takes subset of train\nDEBUG = False\n# Fast Processing\nFAST= False\n# Number of Unique Characters To Predict + Pad Token + SOS Token + EOS Token\nN_UNIQUE_CHARACTERSPAD_TOKEN = len(CHAR2ORD)\nSOS_TOKEN = len(CHAR2ORD) + 1 # Start Of Sentence\nEOS_TOKEN = len(CHAR2ORD) + 2 # End Of Sentence","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:45:36.503972Z","iopub.execute_input":"2023-06-15T11:45:36.504341Z","iopub.status.idle":"2023-06-15T11:45:36.513863Z","shell.execute_reply.started":"2023-06-15T11:45:36.504297Z","shell.execute_reply":"2023-06-15T11:45:36.512677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plot Config","metadata":{}},{"cell_type":"code","source":"# MatplotLib Global Settings\nmpl.rcParams.update(mpl.rcParamsDefault)\nmpl.rcParams['xtick.labelsize'] = 16\nmpl.rcParams['ytick.labelsize'] = 16\nmpl.rcParams['axes.labelsize'] = 18\nmpl.rcParams['axes.titlesize'] = 24","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:45:36.515438Z","iopub.execute_input":"2023-06-15T11:45:36.515765Z","iopub.status.idle":"2023-06-15T11:45:36.52652Z","shell.execute_reply.started":"2023-06-15T11:45:36.515739Z","shell.execute_reply":"2023-06-15T11:45:36.525336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utils","metadata":{}},{"cell_type":"code","source":"# Prints Shape and Dtype For List Of Variables\ndef print_shape_dtype(l, names):\n    for e, n in zip(l, names):\n        print(f'{n} shape: {e.shape}, dtype: {e.dtype}')","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:45:36.527872Z","iopub.execute_input":"2023-06-15T11:45:36.528259Z","iopub.status.idle":"2023-06-15T11:45:36.537288Z","shell.execute_reply.started":"2023-06-15T11:45:36.528221Z","shell.execute_reply":"2023-06-15T11:45:36.536208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read Train","metadata":{}},{"cell_type":"code","source":"# Read Train DataFrame\nif DEBUG:\n    train = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv').head(5000)\nelse:\n    train = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\n\n# Number Of Train Samples\nN_SAMPLES = len(train)\nprint(f'N_SAMPLES: {N_SAMPLES}')\n\ndisplay(train.info())\ndisplay(train.head())","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:45:36.538563Z","iopub.execute_input":"2023-06-15T11:45:36.538881Z","iopub.status.idle":"2023-06-15T11:45:36.747773Z","shell.execute_reply.started":"2023-06-15T11:45:36.538856Z","shell.execute_reply":"2023-06-15T11:45:36.746673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Phrase Type","metadata":{}},{"cell_type":"code","source":"\"\"\"\nAttempt to retrieve phrase type\nCould be used for pretraining or type specific inference\n *) Phone Number\\\n *) URL\n *3) Addres\n\"\"\"\ndef get_phrase_type(phrase):\n    # Phone Number\n    if re.match(r'^[\\d+-]+$', phrase):\n        return 'phone_number'\n    # url\n    elif any([substr in phrase for substr in ['www', '.', '/']]) and ' ' not in phrase:\n        return 'url'\n    # Address\n    else:\n        return 'address'\n    \ntrain['phrase_type'] = train['phrase'].apply(get_phrase_type)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:45:36.749229Z","iopub.execute_input":"2023-06-15T11:45:36.749633Z","iopub.status.idle":"2023-06-15T11:45:36.877356Z","shell.execute_reply.started":"2023-06-15T11:45:36.749595Z","shell.execute_reply":"2023-06-15T11:45:36.876271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Add File Path","metadata":{}},{"cell_type":"code","source":"# Get complete file path to file\ndef get_file_path(path):\n    return f'/kaggle/input/asl-fingerspelling/{path}'\n\ntrain['file_path'] = train['path'].apply(get_file_path)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:45:36.878856Z","iopub.execute_input":"2023-06-15T11:45:36.879747Z","iopub.status.idle":"2023-06-15T11:45:36.907465Z","shell.execute_reply.started":"2023-06-15T11:45:36.879708Z","shell.execute_reply":"2023-06-15T11:45:36.90654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Phrase Processing","metadata":{}},{"cell_type":"code","source":"# Split Phrase To Char Tuple\ntrain['phrase_char'] = train['phrase'].apply(tuple)\n# Character Length of Phrase\ntrain['phrase_char_len'] = train['phrase_char'].apply(len)\n\n# Maximum Input Length\nMAX_PHRASE_LENGTH = train['phrase_char_len'].max()\nprint(f'MAX_PHRASE_LENGTH: {MAX_PHRASE_LENGTH}')\n\n# Train DataFrame indexed by sequence_id to convenientlyy lookup recording data\ntrain_sequence_id = train.set_index('sequence_id')","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:45:36.912514Z","iopub.execute_input":"2023-06-15T11:45:36.912859Z","iopub.status.idle":"2023-06-15T11:45:37.017047Z","shell.execute_reply.started":"2023-06-15T11:45:36.912831Z","shell.execute_reply":"2023-06-15T11:45:37.01576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Phrase Character Length Statistics\ndisplay(train['phrase_char_len'].describe(percentiles=PERCENTILES).to_frame().round(1))","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:45:37.018673Z","iopub.execute_input":"2023-06-15T11:45:37.019243Z","iopub.status.idle":"2023-06-15T11:45:37.04206Z","shell.execute_reply.started":"2023-06-15T11:45:37.01921Z","shell.execute_reply":"2023-06-15T11:45:37.041033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Character Count Occurance\nplt.figure(figsize=(15,8))\nplt.title('Character Length Occurance of Phrases')\ntrain['phrase_char_len'].value_counts().sort_index().plot(kind='bar')\nplt.xlim(-0.50, train['phrase_char_len'].max() - 1.50)\nplt.xlabel('Pharse Character Length')\nplt.ylabel('Sample Count')\nplt.grid(axis='y')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:45:37.043511Z","iopub.execute_input":"2023-06-15T11:45:37.044095Z","iopub.status.idle":"2023-06-15T11:45:37.554252Z","shell.execute_reply.started":"2023-06-15T11:45:37.044056Z","shell.execute_reply":"2023-06-15T11:45:37.553234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Find Unique Character","metadata":{}},{"cell_type":"code","source":"# Use Set to keep track of unique characters in phrases\nUNIQUE_CHARACTERS = set()\n\nfor phrase in tqdm(train['phrase_char']):\n    for c in phrase:\n        UNIQUE_CHARACTERS.add(c)\n        \n# Sorted Unique Character\nUNIQUE_CHARACTERS = np.array(sorted(UNIQUE_CHARACTERS))\n# Number of Unique Characters\nN_UNIQUE_CHARACTERS = len(UNIQUE_CHARACTERS)\nprint(f'N_UNIQUE_CHARACTERS: {N_UNIQUE_CHARACTERS}')","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:45:37.55579Z","iopub.execute_input":"2023-06-15T11:45:37.55672Z","iopub.status.idle":"2023-06-15T11:45:37.770726Z","shell.execute_reply.started":"2023-06-15T11:45:37.556682Z","shell.execute_reply":"2023-06-15T11:45:37.769604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Example Parquet File","metadata":{}},{"cell_type":"code","source":"# Read First Parquet File\nexample_parquet_df = pd.read_parquet(train['file_path'][0])\n\n# Each parquet file contains 1000 recordings\nprint(f'# Unique Recording: {example_parquet_df.index.nunique()}')\n# Display DataFrame layout\ndisplay(example_parquet_df.head())","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:45:37.77267Z","iopub.execute_input":"2023-06-15T11:45:37.773089Z","iopub.status.idle":"2023-06-15T11:45:52.62699Z","shell.execute_reply.started":"2023-06-15T11:45:37.773051Z","shell.execute_reply":"2023-06-15T11:45:52.625982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Video Statistics","metadata":{}},{"cell_type":"code","source":"# Number of parquet chunks to analyse\nN = 5 if IS_INTERACTIVE else 25\n# Number of Unique Frames in Recording\nN_UNIQUE_FRAMES = []\n\nUNIQUE_FILE_PATHS = pd.Series(train['file_path'].unique())\n\nfor idx, file_path in enumerate(tqdm(UNIQUE_FILE_PATHS.sample(N, random_state=SEED))):\n    df = pd.read_parquet(file_path)\n    for group, group_df in df.groupby('sequence_id'):\n        N_UNIQUE_FRAMES.append(group_df['frame'].nunique())\n\n# Convert to Numpy Array\nN_UNIQUE_FRAMES = np.array(N_UNIQUE_FRAMES)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:45:52.628228Z","iopub.execute_input":"2023-06-15T11:45:52.628544Z","iopub.status.idle":"2023-06-15T11:47:14.30509Z","shell.execute_reply.started":"2023-06-15T11:45:52.628516Z","shell.execute_reply":"2023-06-15T11:47:14.303602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of unique frames in each video\ndisplay(pd.Series(N_UNIQUE_FRAMES).describe(percentiles=PERCENTILES).to_frame('Value').astype(int))\n\nplt.figure(figsize=(15,8))\nplt.title('Number of Unique Frames', size=24)\npd.Series(N_UNIQUE_FRAMES).plot(kind='hist', bins=128)\nplt.grid()\nxlim = math.ceil(plt.xlim()[1])\nplt.xlim(0, xlim)\nplt.xticks(np.arange(0, xlim+50, 50))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:47:14.307134Z","iopub.execute_input":"2023-06-15T11:47:14.308775Z","iopub.status.idle":"2023-06-15T11:47:15.240206Z","shell.execute_reply.started":"2023-06-15T11:47:14.308676Z","shell.execute_reply":"2023-06-15T11:47:15.238964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# With N_TARGET_FRAMES = 256 ~85% will be below\nN_UNIQUE_FRAMES_WATERFALL = []\n# Maximum Number of Unique Frames to use\nN_MAX_UNIQUE_FRAMES = 400\n# Compute Percentage\nfor n in tqdm(range(0,N_MAX_UNIQUE_FRAMES+1)):\n    N_UNIQUE_FRAMES_WATERFALL.append(sum(N_UNIQUE_FRAMES >= n) / len(N_UNIQUE_FRAMES) * 100)\n\nplt.figure(figsize=(18,10))\nplt.title('Waterfall Plot For Number Of Unique Frames')\npd.Series(N_UNIQUE_FRAMES_WATERFALL).plot(kind='bar')\nplt.grid(axis='y')\nplt.xticks([1] + np.arange(5, N_MAX_UNIQUE_FRAMES+5, 5).tolist(), size=8, rotation=45)\nplt.xlabel('Number of Unique Frames', size=16)\nplt.yticks(np.arange(0, 100+5, 5), [f'{i}%' for i in range(0,100+5,5)])\nplt.ylim(0, 100)\nplt.ylabel('Percentage of Samples With At Least N Unique Frames', size=16)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:47:15.242032Z","iopub.execute_input":"2023-06-15T11:47:15.242853Z","iopub.status.idle":"2023-06-15T11:47:17.205541Z","shell.execute_reply.started":"2023-06-15T11:47:15.24281Z","shell.execute_reply":"2023-06-15T11:47:17.204275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Landmark Indices","metadata":{}},{"cell_type":"code","source":"def get_idxs(df, words_pos, words_neg=[], ret_names=True, idxs_pos=None):\n    idxs = []\n    names = []\n    for w in words_pos:\n        for col_idx, col in enumerate(example_parquet_df.columns):\n            # Exclude Non Landmark Columns\n            if col in ['frame']:\n                continue\n                \n            col_idx = int(col.split('_')[-1])\n            # Check if column name contains all words\n            if (w in col) and (idxs_pos is None or col_idx in idxs_pos) and all([w not in col for w in words_neg]):\n                idxs.append(col_idx)\n                names.append(col)\n    # Convert to Numpy arrays\n    idxs = np.array(idxs)\n    names = np.array(names)\n    # Returns either both column indices and names\n    if ret_names:\n        return idxs, names\n    # Or only columns indices\n    else:\n        return idxs","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:47:17.207055Z","iopub.execute_input":"2023-06-15T11:47:17.207634Z","iopub.status.idle":"2023-06-15T11:47:17.217667Z","shell.execute_reply.started":"2023-06-15T11:47:17.207599Z","shell.execute_reply":"2023-06-15T11:47:17.21585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lips Landmark Face Ids\nLIPS_LANDMARK_IDXS = np.array([\n        61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n        291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n        78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n        95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n    ])\n\n# Landmark Indices for Left/Right hand without z axis in raw data\nLEFT_HAND_IDXS0, LEFT_HAND_NAMES0 = get_idxs(example_parquet_df, ['left_hand'], ['z'])\nRIGHT_HAND_IDXS0, RIGHT_HAND_NAMES0 = get_idxs(example_parquet_df, ['right_hand'], ['z'])\nLIPS_IDXS0, LIPS_NAMES0 = get_idxs(example_parquet_df, ['face'], ['z'], idxs_pos=LIPS_LANDMARK_IDXS)\nCOLUMNS0 = np.concatenate((LEFT_HAND_NAMES0, RIGHT_HAND_NAMES0, LIPS_NAMES0))\nN_COLS0 = len(COLUMNS0)\n# Only X/Y axes are used\nN_DIMS0 = 2\n\nprint(f'N_COLS0: {N_COLS0}')","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:47:17.218805Z","iopub.execute_input":"2023-06-15T11:47:17.219156Z","iopub.status.idle":"2023-06-15T11:47:17.239738Z","shell.execute_reply.started":"2023-06-15T11:47:17.219129Z","shell.execute_reply":"2023-06-15T11:47:17.23847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Landmark Indices in subset of dataframe with only COLUMNS selected\nLEFT_HAND_IDXS = np.argwhere(np.isin(COLUMNS0, LEFT_HAND_NAMES0)).squeeze()\nRIGHT_HAND_IDXS = np.argwhere(np.isin(COLUMNS0, RIGHT_HAND_NAMES0)).squeeze()\nLIPS_IDXS = np.argwhere(np.isin(COLUMNS0, LIPS_NAMES0)).squeeze()\nN_COLS = N_COLS0\n# Only X/Y axes are used\nN_DIMS = 2\n\nprint(f'N_COLS: {N_COLS}')","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:47:17.24125Z","iopub.execute_input":"2023-06-15T11:47:17.241562Z","iopub.status.idle":"2023-06-15T11:47:17.260947Z","shell.execute_reply.started":"2023-06-15T11:47:17.241537Z","shell.execute_reply":"2023-06-15T11:47:17.259506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Indices in processed data by axes with only dominant hand\nHAND_X_IDXS = np.array(\n        [idx for idx, name in enumerate(LEFT_HAND_NAMES0) if 'x' in name]\n    ).squeeze()\nHAND_Y_IDXS = np.array(\n        [idx for idx, name in enumerate(LEFT_HAND_NAMES0) if 'y' in name]\n    ).squeeze()\n# Names in processed data by axes\nHAND_X_NAMES = LEFT_HAND_NAMES0[HAND_X_IDXS]\nHAND_Y_NAMES = LEFT_HAND_NAMES0[HAND_Y_IDXS]","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:47:17.262873Z","iopub.execute_input":"2023-06-15T11:47:17.263524Z","iopub.status.idle":"2023-06-15T11:47:17.269434Z","shell.execute_reply.started":"2023-06-15T11:47:17.263486Z","shell.execute_reply":"2023-06-15T11:47:17.268426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Number Of Non-NaN Frames","metadata":{}},{"cell_type":"code","source":"\"\"\"\n    Tensorflow layer to process data in TFLite\n    Data needs to be processed in the model itself, so we can not use Python\n\"\"\" \nclass PreprocessLayerNonNaN(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayerNonNaN, self).__init__()\n    \n    @tf.function(\n        input_signature=(tf.TensorSpec(shape=[None,N_COLS0], dtype=tf.float32),),\n    )\n    def call(self, data0):\n        # Fill NaN Values With 0\n        data = tf.where(tf.math.is_nan(data0), 0.0, data0)\n        \n        # Hacky\n        data = data[None]\n        \n        # Empty Hand Frame Filtering\n        hands = tf.slice(data, [0,0,0], [-1, -1, 84])\n        hands = tf.abs(hands)\n        mask = tf.reduce_sum(hands, axis=2)\n        mask = tf.not_equal(mask, 0)\n        data = data[mask][None]\n        data = tf.squeeze(data, axis=[0])\n        \n        return data\n    \npreprocess_layer_non_nan = PreprocessLayerNonNaN()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:47:17.270557Z","iopub.execute_input":"2023-06-15T11:47:17.271338Z","iopub.status.idle":"2023-06-15T11:47:17.302082Z","shell.execute_reply.started":"2023-06-15T11:47:17.271307Z","shell.execute_reply":"2023-06-15T11:47:17.300702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Unique Parquet Files\nUNIQUE_FILE_PATHS = pd.Series(train['file_path'].unique())\n# Number of parquet chunks to analyse\nN = 5 if (IS_INTERACTIVE or FAST) else len(UNIQUE_FILE_PATHS)\n# Number of Non Nan Frames in Recording\nN_NON_NAN_FRAMES = []\n\nfor idx, file_path in enumerate(tqdm(UNIQUE_FILE_PATHS.sample(N, random_state=SEED))):\n    df = pd.read_parquet(file_path)\n    for group, group_df in df.groupby('sequence_id'):\n        frames = preprocess_layer_non_nan(group_df[COLUMNS0].values).numpy()\n        N_NON_NAN_FRAMES.append(len(frames))\n\n# Convert to Numpy Array\nN_NON_NAN_FRAMES = pd.Series(N_NON_NAN_FRAMES).to_frame('# Frames')","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:47:17.303624Z","iopub.execute_input":"2023-06-15T11:47:17.303947Z","iopub.status.idle":"2023-06-15T11:47:56.769462Z","shell.execute_reply.started":"2023-06-15T11:47:17.303918Z","shell.execute_reply":"2023-06-15T11:47:56.768271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of frames in each video with hand coordinates\ndisplay(N_NON_NAN_FRAMES.describe(percentiles=PERCENTILES).astype(int))\n\nN_NON_NAN_FRAMES.plot(kind='hist', bins=128, figsize=(15,8))\nplt.title('Number of Non NaN Frames', size=24)\nplt.grid()\nxlim = np.percentile(N_NON_NAN_FRAMES, 99)\nplt.xlim(0, xlim)\nplt.xticks(np.arange(0, xlim+32, 32))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:47:56.77075Z","iopub.execute_input":"2023-06-15T11:47:56.77109Z","iopub.status.idle":"2023-06-15T11:47:57.305762Z","shell.execute_reply.started":"2023-06-15T11:47:56.771059Z","shell.execute_reply":"2023-06-15T11:47:57.304664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tensorflow Preprocess Layer","metadata":{}},{"cell_type":"code","source":"\"\"\"\n    Tensorflow layer to process data in TFLite\n    Data needs to be processed in the model itself, so we can not use Python\n\"\"\" \nclass PreprocessLayer(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayer, self).__init__()\n    \n    @tf.function(\n        input_signature=(tf.TensorSpec(shape=[None,N_COLS0], dtype=tf.float32),),\n    )\n    def call(self, data0, resize=True):\n        # Fill NaN Values With 0\n        data = tf.where(tf.math.is_nan(data0), 0.0, data0)\n        \n        # Hacky\n        data = data[None]\n        \n        # Empty Hand Frame Filtering\n        hands = tf.slice(data, [0,0,0], [-1, -1, 84])\n        hands = tf.abs(hands)\n        mask = tf.reduce_sum(hands, axis=2)\n        mask = tf.not_equal(mask, 0)\n        data = data[mask][None]\n        \n        # Pad Zeros\n        N_FRAMES = len(data[0])\n        if N_FRAMES < N_TARGET_FRAMES:\n            data = tf.concat((\n                data,\n                tf.zeros([1,N_TARGET_FRAMES-N_FRAMES,N_COLS], dtype=tf.float32)\n            ), axis=1)\n        # Downsample\n        data = tf.image.resize(\n            data,\n            [1, N_TARGET_FRAMES],\n            method=tf.image.ResizeMethod.BILINEAR,\n        )\n        \n        # Squeeze Batch Dimension\n        data = tf.squeeze(data, axis=[0])\n        \n        return data\n    \npreprocess_layer = PreprocessLayer()\n\ninputs = group_df[COLUMNS0].values\ninputs = inputs[:1]\n\nframes = preprocess_layer(inputs)\n\nprint(f'inputs shape: {inputs.shape}')\nprint(f'frames shape: {frames.shape}, NaN count: {np.isnan(frames).sum()}')","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:47:57.307673Z","iopub.execute_input":"2023-06-15T11:47:57.3081Z","iopub.status.idle":"2023-06-15T11:47:57.576595Z","shell.execute_reply.started":"2023-06-15T11:47:57.308062Z","shell.execute_reply":"2023-06-15T11:47:57.575437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create X/Y","metadata":{}},{"cell_type":"code","source":"# Target Arrays Processed Input Videos\nX = np.zeros([N_SAMPLES, N_TARGET_FRAMES, N_COLS], dtype=np.float32)\n# Ordinally Encoded Target With value 59 for pad token\ny = np.full(shape=[N_SAMPLES, N_TARGET_FRAMES], fill_value=N_UNIQUE_CHARACTERS, dtype=np.int8)\n# Phrase Type\ny_phrase_type = np.empty(shape=[N_SAMPLES], dtype=object)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:47:57.578391Z","iopub.execute_input":"2023-06-15T11:47:57.578932Z","iopub.status.idle":"2023-06-15T11:47:57.587556Z","shell.execute_reply.started":"2023-06-15T11:47:57.578892Z","shell.execute_reply":"2023-06-15T11:47:57.586553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# All Unique Parquet Files\nUNIQUE_FILE_PATHS = pd.Series(train['file_path'].unique())\nN_UNIQUE_FILE_PATHS = len(UNIQUE_FILE_PATHS)\n# Counter to keep track of sample\nrow = 0\ncount = 0\n# Compressed Parquet Files\nPath('train_landmark_subsets').mkdir(parents=True, exist_ok=True)\n# Numbre Of Frames Per Character\nN_FRAMES_PER_CHARACTER = []\n# Minimum Number Of Frames Per Character\nMIN_NUM_FRAMES_PER_CHARACTER = 4\nVALID_IDXS = []\n\n# Fill Arrays\nfor idx, file_path in enumerate(tqdm(UNIQUE_FILE_PATHS)):\n    # Progress Logging\n    print(f'Processed {idx:02d}/{N_UNIQUE_FILE_PATHS} parquet files')\n    # Read parquet file\n    df = pd.read_parquet(file_path)\n    # Save COLUMN Subset of parquet files for TFLite Model verficiation\n    name = file_path.split('/')[-1]\n    if idx < 10:\n        df[COLUMNS0].to_parquet(f'train_landmark_subsets/{name}', engine='pyarrow', compression='zstd')\n    # Iterate Over Samples\n    for group, group_df in df.groupby('sequence_id'):\n        # Number of Frames Per Character\n        n_frames_per_character =  len(group_df[COLUMNS0].values) / len(train_sequence_id.loc[group, 'phrase_char'])\n        N_FRAMES_PER_CHARACTER.append(n_frames_per_character)\n        if n_frames_per_character < MIN_NUM_FRAMES_PER_CHARACTER:\n            count = count + 1\n            continue\n        else:\n            # Add Valid Index\n            VALID_IDXS.append(count)\n            count = count + 1\n        \n        # Get Processed Frames and non empty frame indices\n        frames = preprocess_layer(group_df[COLUMNS0].values)\n        assert frames.ndim == 2\n        # Assign\n        X[row] = frames\n        # Add Target By Ordinally Encoding Characters\n        phrase_char = train_sequence_id.loc[group, 'phrase_char']\n        for col, char in enumerate(phrase_char):\n            y[row, col] = CHAR2ORD.get(char)\n        # Add EOS Token\n        y[row, col+1] = EOS_TOKEN\n        # Phrase Type\n        y_phrase_type[row] = train_sequence_id.loc[group, 'phrase_type']\n        # Row Count\n        row += 1\n    # clean up\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T11:47:57.589002Z","iopub.execute_input":"2023-06-15T11:47:57.589895Z","iopub.status.idle":"2023-06-15T12:09:53.471933Z","shell.execute_reply.started":"2023-06-15T11:47:57.589854Z","shell.execute_reply":"2023-06-15T12:09:53.470759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rows denotes the number of samples with frames/character above threshold\nprint(f'row: {row}, count: {count}')","metadata":{"execution":{"iopub.status.busy":"2023-06-15T12:09:53.480049Z","iopub.execute_input":"2023-06-15T12:09:53.481056Z","iopub.status.idle":"2023-06-15T12:09:53.486529Z","shell.execute_reply.started":"2023-06-15T12:09:53.481019Z","shell.execute_reply":"2023-06-15T12:09:53.485223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example target, note the phrase is padded with the pad token 59\nprint(f'Example Target: {y[0]}')","metadata":{"execution":{"iopub.status.busy":"2023-06-15T12:09:53.487927Z","iopub.execute_input":"2023-06-15T12:09:53.488387Z","iopub.status.idle":"2023-06-15T12:09:53.513296Z","shell.execute_reply.started":"2023-06-15T12:09:53.488348Z","shell.execute_reply":"2023-06-15T12:09:53.512172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filer X/y\nX = X[:row]\ny = y[:row]","metadata":{"execution":{"iopub.status.busy":"2023-06-15T12:09:53.514816Z","iopub.execute_input":"2023-06-15T12:09:53.515271Z","iopub.status.idle":"2023-06-15T12:09:53.523659Z","shell.execute_reply.started":"2023-06-15T12:09:53.515231Z","shell.execute_reply":"2023-06-15T12:09:53.52284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save X/y\nnp.save('X.npy', X)\nnp.save('y.npy', y)\n# Save Validation\nsplitter = GroupShuffleSplit(test_size=0.10, n_splits=2, random_state=SEED)\nPARTICIPANT_IDS = train['participant_id'].values[VALID_IDXS]\ntrain_idxs, val_idxs = next(splitter.split(X, y, groups=PARTICIPANT_IDS))\n\n# Save Train\nnp.save('X_train.npy', X[train_idxs])\nnp.save('y_train.npy', y[train_idxs])\n# Save Validation\nnp.save('X_val.npy', X[val_idxs])\nnp.save('y_val.npy', y[val_idxs])\n# Verify Train/Val is correctly split by participan id\nprint(f'Patient ID Intersection Train/Val: {set(PARTICIPANT_IDS[train_idxs]).intersection(PARTICIPANT_IDS[val_idxs])}')\n# Train/Val Sizes\nprint(f'# Train Samples: {len(train_idxs)}, # Val Samples: {len(val_idxs)}')","metadata":{"execution":{"iopub.status.busy":"2023-06-15T12:09:53.52518Z","iopub.execute_input":"2023-06-15T12:09:53.525885Z","iopub.status.idle":"2023-06-15T12:10:32.032466Z","shell.execute_reply.started":"2023-06-15T12:09:53.525846Z","shell.execute_reply":"2023-06-15T12:10:32.031128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Number of Frames Per Character","metadata":{}},{"cell_type":"code","source":"N_FRAMES_PER_CHARACTER_S = pd.Series(N_FRAMES_PER_CHARACTER)\n\ndisplay(N_FRAMES_PER_CHARACTER_S.describe(percentiles=PERCENTILES).to_frame('Value').round(2))\n\nplt.figure(figsize=(20,10))\nplt.title('Number Of Frames Per Phrase Character')\nN_FRAMES_PER_CHARACTER_S.plot(kind='hist', bins=128)\n# Plot till 99th percentile\np99 = math.ceil(np.percentile(N_FRAMES_PER_CHARACTER_S, 99))\nplt.xticks(np.arange(0, p99+1, 1))\nplt.xlim(0, p99)\nplt.xlabel('Number Of Frames Per Phrase Character')\nplt.ylabel('Sample Count')\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T12:13:22.972117Z","iopub.execute_input":"2023-06-15T12:13:22.972585Z","iopub.status.idle":"2023-06-15T12:13:23.562121Z","shell.execute_reply.started":"2023-06-15T12:13:22.972548Z","shell.execute_reply":"2023-06-15T12:13:23.56128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Coordinate Statistics","metadata":{}},{"cell_type":"code","source":"def get_left_right_hand_mean_std():\n    # Dominant Hand Statistics\n    MEANS = np.zeros([N_COLS], dtype=np.float32)\n    STDS = np.zeros([N_COLS], dtype=np.float32)\n    \n    # Plot\n    fig, axes = plt.subplots(3, figsize=(20, 3*8))\n    \n    # Iterate over all landmarks\n    for col, v in enumerate(tqdm(X.reshape([-1, N_COLS]).T)):\n        v = v[np.nonzero(v)]\n        # Remove zero values as they are NaN values\n        MEANS[col] = v.astype(np.float32).mean()\n        STDS[col] = v.astype(np.float32).std()\n        if col in LEFT_HAND_IDXS:\n            axes[0].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n        elif col in RIGHT_HAND_IDXS:\n            axes[1].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n        else:\n            axes[2].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n        \n    for ax, name in zip(axes, ['Left Hand', 'Right Hand', 'Lips']):\n        ax.set_title(f'{name}', size=24)\n        ax.tick_params(axis='x', labelsize=8, rotation=45)\n        ax.set_ylim(0.0, 1.0)\n        ax.grid(axis='y')\n\n    plt.show()\n    \n    return MEANS, STDS\n\n# Get Dominant Hand Mean/Standard Deviation\nMEANS, STDS = get_left_right_hand_mean_std()\n# Save Mean/STD to normalize input in neural network model\nnp.save('MEANS.npy', MEANS)\nnp.save('STDS.npy', STDS)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T12:10:32.807334Z","iopub.execute_input":"2023-06-15T12:10:32.808486Z","iopub.status.idle":"2023-06-15T12:12:09.900914Z","shell.execute_reply.started":"2023-06-15T12:10:32.80844Z","shell.execute_reply":"2023-06-15T12:12:09.899813Z"},"trusted":true},"execution_count":null,"outputs":[]}]}