{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install celluloid","metadata":{"execution":{"iopub.status.busy":"2023-07-05T10:59:33.916512Z","iopub.execute_input":"2023-07-05T10:59:33.917621Z","iopub.status.idle":"2023-07-05T10:59:49.374601Z","shell.execute_reply.started":"2023-07-05T10:59:33.917569Z","shell.execute_reply":"2023-07-05T10:59:49.373083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport sklearn\nimport matplotlib.pyplot as plt\nimport json\nimport regex\nimport os\nfrom tqdm.notebook import tqdm\nimport math\nfrom IPython.display import HTML, clear_output\nfrom celluloid import Camera\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-05T10:59:49.37684Z","iopub.execute_input":"2023-07-05T10:59:49.377262Z","iopub.status.idle":"2023-07-05T10:59:50.139805Z","shell.execute_reply.started":"2023-07-05T10:59:49.377223Z","shell.execute_reply":"2023-07-05T10:59:50.138853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# MatplotLib Global Settings\nplt.rcParams.update(plt.rcParamsDefault)\nplt.rcParams['xtick.labelsize'] = 16\nplt.rcParams['ytick.labelsize'] = 16\nplt.rcParams['axes.labelsize'] = 18\nplt.rcParams['axes.titlesize'] = 24","metadata":{"execution":{"iopub.status.busy":"2023-07-05T10:59:50.141289Z","iopub.execute_input":"2023-07-05T10:59:50.141692Z","iopub.status.idle":"2023-07-05T10:59:50.149995Z","shell.execute_reply.started":"2023-07-05T10:59:50.141655Z","shell.execute_reply":"2023-07-05T10:59:50.148888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read Character to Ordinal Encoding Mapping\nwith open('/kaggle/input/asl-fingerspelling/character_to_prediction_index.json') as json_file:\n    char2ord = json.load(json_file)\n\ndisplay(pd.Series(char2ord).to_frame('Ordinal Encoding'))","metadata":{"execution":{"iopub.status.busy":"2023-07-05T10:59:50.153306Z","iopub.execute_input":"2023-07-05T10:59:50.154021Z","iopub.status.idle":"2023-07-05T10:59:50.205706Z","shell.execute_reply.started":"2023-07-05T10:59:50.153978Z","shell.execute_reply":"2023-07-05T10:59:50.204897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# If Notebook Is Run By Committing or In Interactive Mode For Development\nIS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\n\nPERCENTILES = [0.0, 0.1, 0.25, 0.5, 0.75, 0.97, 0.99]\n\nSEED = 42\n\nno_unique_char_tokens = len(char2ord)\nSOS_TOKEN = len(char2ord) + 1\nEOS_TOKEN = len(char2ord) + 2","metadata":{"execution":{"iopub.status.busy":"2023-07-05T10:59:50.206789Z","iopub.execute_input":"2023-07-05T10:59:50.207866Z","iopub.status.idle":"2023-07-05T10:59:50.213898Z","shell.execute_reply.started":"2023-07-05T10:59:50.207824Z","shell.execute_reply":"2023-07-05T10:59:50.213146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\ndisplay(train_df.head())\nprint('Number of samples:', len(train_df))","metadata":{"execution":{"iopub.status.busy":"2023-07-05T10:59:50.215233Z","iopub.execute_input":"2023-07-05T10:59:50.215914Z","iopub.status.idle":"2023-07-05T10:59:50.390305Z","shell.execute_reply.started":"2023-07-05T10:59:50.215883Z","shell.execute_reply":"2023-07-05T10:59:50.389163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.info())","metadata":{"execution":{"iopub.status.busy":"2023-07-05T10:59:50.391649Z","iopub.execute_input":"2023-07-05T10:59:50.391974Z","iopub.status.idle":"2023-07-05T10:59:50.46067Z","shell.execute_reply.started":"2023-07-05T10:59:50.391947Z","shell.execute_reply":"2023-07-05T10:59:50.459615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"No empty rows","metadata":{}},{"cell_type":"code","source":"train_df['phrase_len'] = train_df['phrase'].apply(len)\nmax_phrase_len = train_df['phrase_len'].max()\nprint('Maximum phrase length:', max_phrase_len)","metadata":{"execution":{"iopub.status.busy":"2023-07-05T10:59:50.462091Z","iopub.execute_input":"2023-07-05T10:59:50.462466Z","iopub.status.idle":"2023-07-05T10:59:50.509975Z","shell.execute_reply.started":"2023-07-05T10:59:50.462436Z","shell.execute_reply":"2023-07-05T10:59:50.508815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequence_id = train_df.set_index('sequence_id')\ndisplay(sequence_id)","metadata":{"execution":{"iopub.status.busy":"2023-07-05T10:59:50.511924Z","iopub.execute_input":"2023-07-05T10:59:50.512447Z","iopub.status.idle":"2023-07-05T10:59:50.535226Z","shell.execute_reply.started":"2023-07-05T10:59:50.512407Z","shell.execute_reply":"2023-07-05T10:59:50.534186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number of parquet files:',len(np.unique(train_df['path'])))","metadata":{"execution":{"iopub.status.busy":"2023-07-05T10:59:50.540927Z","iopub.execute_input":"2023-07-05T10:59:50.541316Z","iopub.status.idle":"2023-07-05T10:59:50.622192Z","shell.execute_reply.started":"2023-07-05T10:59:50.541284Z","shell.execute_reply":"2023-07-05T10:59:50.621102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def file_paths(path):\n    return f'/kaggle/input/asl-fingerspelling/{path}'\ntrain_df['file_paths'] = train_df['path'].apply(file_paths)\ndisplay(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-07-05T10:59:50.623627Z","iopub.execute_input":"2023-07-05T10:59:50.624065Z","iopub.status.idle":"2023-07-05T10:59:50.676099Z","shell.execute_reply.started":"2023-07-05T10:59:50.624025Z","shell.execute_reply":"2023-07-05T10:59:50.675321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analysis of Phrases","metadata":{}},{"cell_type":"code","source":"# Character Count Occurence\nplt.figure(figsize=(15,8))\nplt.title('Character Length Occurence of Phrases')\ntrain_df['phrase_len'].value_counts().sort_index().plot(kind='bar')\nplt.xlim(-0.50, train_df['phrase_len'].max() - 1.50)\nplt.xlabel('Phrase Character Length')\nplt.ylabel('Sample Count')\nplt.grid(axis='y')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-05T10:59:50.677143Z","iopub.execute_input":"2023-07-05T10:59:50.677684Z","iopub.status.idle":"2023-07-05T10:59:51.321986Z","shell.execute_reply.started":"2023-07-05T10:59:50.677654Z","shell.execute_reply":"2023-07-05T10:59:51.320223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Phrase Character Length Statistics\ndisplay(train_df['phrase_len'].describe(percentiles=PERCENTILES).to_frame().round(1))\nprint('Maximum phrase length:', train_df['phrase_len'].max())","metadata":{"execution":{"iopub.status.busy":"2023-07-05T10:59:51.323879Z","iopub.execute_input":"2023-07-05T10:59:51.326448Z","iopub.status.idle":"2023-07-05T10:59:51.359411Z","shell.execute_reply.started":"2023-07-05T10:59:51.326402Z","shell.execute_reply":"2023-07-05T10:59:51.357621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analysis of Sequences and frames","metadata":{}},{"cell_type":"code","source":"landmark_df = pd.read_parquet(train_df['file_paths'][0])\ndisplay(landmark_df)","metadata":{"execution":{"iopub.status.busy":"2023-07-05T10:59:51.361375Z","iopub.execute_input":"2023-07-05T10:59:51.36198Z","iopub.status.idle":"2023-07-05T11:00:05.303259Z","shell.execute_reply.started":"2023-07-05T10:59:51.361947Z","shell.execute_reply":"2023-07-05T11:00:05.301923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"landmark_df.columns","metadata":{"execution":{"iopub.status.busy":"2023-07-05T11:00:05.304559Z","iopub.execute_input":"2023-07-05T11:00:05.304894Z","iopub.status.idle":"2023-07-05T11:00:05.313304Z","shell.execute_reply.started":"2023-07-05T11:00:05.304866Z","shell.execute_reply":"2023-07-05T11:00:05.312026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of parquet chunks to analyse\nN = 5 if IS_INTERACTIVE else 25\n\nN_UNIQUE_FRAMES = []\nUNIQUE_FILE_PATHS = pd.Series(train_df['file_paths'].unique())\n\nfor file in UNIQUE_FILE_PATHS.sample(N, random_state=SEED):\n    df = pd.read_parquet(file)\n    for group, group_df in landmark_df.groupby(by='sequence_id'):\n        N_UNIQUE_FRAMES.append(group_df['frame'].nunique())\n\n# Convert to Numpy Array\nN_UNIQUE_FRAMES = np.array(N_UNIQUE_FRAMES)","metadata":{"execution":{"iopub.status.busy":"2023-07-05T11:00:05.315255Z","iopub.execute_input":"2023-07-05T11:00:05.316071Z","iopub.status.idle":"2023-07-05T11:01:18.58391Z","shell.execute_reply.started":"2023-07-05T11:00:05.316036Z","shell.execute_reply":"2023-07-05T11:01:18.582787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of unique frames in each video\ndisplay(pd.Series(N_UNIQUE_FRAMES).describe(percentiles=PERCENTILES).to_frame('Value').astype(int))\n\nplt.figure(figsize=(15,8))\nplt.title('Number of Unique Frames', size=24)\npd.Series(N_UNIQUE_FRAMES).plot(kind='hist', bins=128)\nplt.grid()\nxlim = math.ceil(plt.xlim()[1])\nplt.xlim(0, xlim)\nplt.xticks(np.arange(0, xlim+50, 50))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-05T11:01:18.58591Z","iopub.execute_input":"2023-07-05T11:01:18.586858Z","iopub.status.idle":"2023-07-05T11:01:19.226311Z","shell.execute_reply.started":"2023-07-05T11:01:18.586809Z","shell.execute_reply":"2023-07-05T11:01:19.225451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_hands(x, y):\n    graphs = [[0,1,2,3,4],\n             [0,5,6,7,8],\n              [5,9,13,17],\n             [9,10,11,12],\n              [13,14,15,16],\n              [0,17,18,19,20]]\n    \n    for g1 in graphs:\n        plt.plot(x[g1], y[g1])\n    #plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-05T11:01:19.22763Z","iopub.execute_input":"2023-07-05T11:01:19.228523Z","iopub.status.idle":"2023-07-05T11:01:19.23562Z","shell.execute_reply.started":"2023-07-05T11:01:19.228489Z","shell.execute_reply":"2023-07-05T11:01:19.233991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_sample(seq_id, n_frames = None):\n    x_test_right = landmark_df.filter(regex = 'x_right_hand')[landmark_df.index == seq_id].dropna()\n    y_test_right = landmark_df.filter(regex = 'y_right_hand')[landmark_df.index == seq_id].dropna()\n    camera = Camera(plt.figure())\n\n    for i in range(x_test_right.shape[0]):\n        x_r = 1 - x_test_right.iloc[i].values\n        y_r = 1 - y_test_right.iloc[i].values\n        plot_hands(x_r, y_r)\n        camera.snap()\n    anim = camera.animate().to_html5_video()\n    return anim","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-07-05T11:01:19.237269Z","iopub.execute_input":"2023-07-05T11:01:19.237649Z","iopub.status.idle":"2023-07-05T11:01:19.249783Z","shell.execute_reply.started":"2023-07-05T11:01:19.237605Z","shell.execute_reply":"2023-07-05T11:01:19.247449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ind = np.random.choice(np.unique(landmark_df.index))\nind = np.unique(landmark_df.index)[0]\nHTML(plot_sample(ind))","metadata":{"execution":{"iopub.status.busy":"2023-07-05T11:01:19.251674Z","iopub.execute_input":"2023-07-05T11:01:19.252467Z","iopub.status.idle":"2023-07-05T11:01:24.032968Z","shell.execute_reply.started":"2023-07-05T11:01:19.252434Z","shell.execute_reply":"2023-07-05T11:01:24.03157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_idxs(df, words_pos, words_neg=[], ret_names=True, idxs_pos=None):\n    idxs = []\n    names = []\n    for w in words_pos:\n        for col_idx, col in enumerate(landmark_df.columns):\n            # Exclude Non Landmark Columns\n            if col in ['frame']:\n                continue\n                \n            col_idx = int(col.split('_')[-1])\n            # Check if column name contains all words\n            if (w in col) and (idxs_pos is None or col_idx in idxs_pos) and all([w not in col for w in words_neg]):\n                idxs.append(col_idx) \n                names.append(col)\n    # Convert to Numpy arrays\n    idxs = np.array(idxs)\n    names = np.array(names)\n    # Returns either both column indices and names\n    if ret_names:\n        return idxs, names\n    # Or only columns indices\n    else:\n        return idxs","metadata":{"execution":{"iopub.status.busy":"2023-07-05T11:01:24.034786Z","iopub.execute_input":"2023-07-05T11:01:24.035518Z","iopub.status.idle":"2023-07-05T11:01:24.046703Z","shell.execute_reply.started":"2023-07-05T11:01:24.035476Z","shell.execute_reply":"2023-07-05T11:01:24.045197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lips Landmark Face Ids\nLIPS_LANDMARK_IDXS = np.array([\n        61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n        291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n        78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n        95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n    ])\n\n# Landmark Indices for Left/Right hand without z axis in raw data\nLEFT_HAND_IDXS0, LEFT_HAND_NAMES0 = get_idxs(landmark_df, ['left_hand'], ['z'])\nRIGHT_HAND_IDXS0, RIGHT_HAND_NAMES0 = get_idxs(landmark_df, ['right_hand'], ['z'])\nLIPS_IDXS0, LIPS_NAMES0 = get_idxs(landmark_df, ['face'], ['z'], idxs_pos=LIPS_LANDMARK_IDXS)\nCOLUMNS0 = np.concatenate((LEFT_HAND_NAMES0, RIGHT_HAND_NAMES0, LIPS_NAMES0))\nN_COLS0 = len(COLUMNS0)\n# Only X/Y axes are used\nN_DIMS0 = 2\n\nprint(f'N_COLS0: {N_COLS0}')","metadata":{"execution":{"iopub.status.busy":"2023-07-05T11:01:24.048384Z","iopub.execute_input":"2023-07-05T11:01:24.048936Z","iopub.status.idle":"2023-07-05T11:01:24.077841Z","shell.execute_reply.started":"2023-07-05T11:01:24.048897Z","shell.execute_reply":"2023-07-05T11:01:24.076597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def hands(landmark_df):\n    MAX_FRAMES = 500\n\n    # Extract x and y coordinate columns of the right hand\n    right_hand_columns = landmark_df.filter(regex='x_right_hand_|y_right_hand_')\n\n    # Extract x and y coordinate columns of the left hand\n    left_hand_columns = landmark_df.filter(regex='x_left_hand_|y_left_hand_')\n\n    # Concatenate the right hand and left hand columns\n    hand_columns = pd.concat([right_hand_columns, left_hand_columns], axis=1).groupby(by='sequence_id')\n\n    for group_name, group_df in hand_columns:\n        if len(group_df)<MAX_FRAMES:\n\n            # Specify the number of empty rows to add\n            num_empty_rows = MAX_FRAMES-len(group_df)\n            new_index = pd.Index([group_name] * num_empty_rows)\n\n            # Create an empty DataFrame with the desired number of rows\n            empty_rows = pd.DataFrame(index=new_index, columns=group_df.columns)\n\n            # Concatenate the original DataFrame with the empty rows DataFrame\n            group_df = pd.concat([group_df, empty_rows])\n        else:\n            group_df = group_df[:MAX_FRAMES]\n        group_df = group_df.fillna(0)\n        result = train_df.loc[train_df['sequence_id'] == group_name, 'phrase'].values[0]\n        sequence_phrases.append(result)\n        frames.append(group_df)","metadata":{"execution":{"iopub.status.busy":"2023-07-05T11:01:24.079381Z","iopub.execute_input":"2023-07-05T11:01:24.079824Z","iopub.status.idle":"2023-07-05T11:01:24.090348Z","shell.execute_reply.started":"2023-07-05T11:01:24.079785Z","shell.execute_reply":"2023-07-05T11:01:24.08904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequence_phrases = []\nframes = []\nfor idx, file in enumerate(tqdm(train_df['file_paths'].unique())):\n    landmark_df = pd.read_parquet(file)\n    hands(landmark_df)\n    clear_output(wait=True)\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-07-05T11:01:24.091576Z","iopub.execute_input":"2023-07-05T11:01:24.091901Z","iopub.status.idle":"2023-07-05T12:39:44.709986Z","shell.execute_reply.started":"2023-07-05T11:01:24.091874Z","shell.execute_reply":"2023-07-05T12:39:44.708468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"display(np.array(frames).shape)\ndisplay(np.array(sequence_phrases).shape)\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-07-05T12:39:44.712642Z","iopub.execute_input":"2023-07-05T12:39:44.713088Z","iopub.status.idle":"2023-07-05T12:39:44.725676Z","shell.execute_reply.started":"2023-07-05T12:39:44.713051Z","shell.execute_reply":"2023-07-05T12:39:44.723662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frames_array = np.array(frames)\nphrase_array = np.array(sequence_phrases)","metadata":{"execution":{"iopub.status.busy":"2023-07-05T12:45:07.836438Z","iopub.execute_input":"2023-07-05T12:45:07.836926Z","iopub.status.idle":"2023-07-05T12:45:45.03532Z","shell.execute_reply.started":"2023-07-05T12:45:07.836895Z","shell.execute_reply":"2023-07-05T12:45:45.033873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-07-05T12:44:41.906814Z","iopub.execute_input":"2023-07-05T12:44:41.90742Z","iopub.status.idle":"2023-07-05T12:44:42.514314Z","shell.execute_reply.started":"2023-07-05T12:44:41.907385Z","shell.execute_reply":"2023-07-05T12:44:42.512623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save('X.npy', frames_array)\nnp.save('Y.npy', phrase_array)","metadata":{"execution":{"iopub.status.busy":"2023-07-05T12:48:45.727054Z","iopub.execute_input":"2023-07-05T12:48:45.72887Z","iopub.status.idle":"2023-07-05T12:49:46.544291Z","shell.execute_reply.started":"2023-07-05T12:48:45.728822Z","shell.execute_reply":"2023-07-05T12:49:46.542401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}