{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport pyarrow.parquet as pq\nfrom IPython.display import display\n\"\"\"\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\"\"\"\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-24T19:13:17.466925Z","iopub.execute_input":"2023-06-24T19:13:17.467338Z","iopub.status.idle":"2023-06-24T19:13:17.475568Z","shell.execute_reply.started":"2023-06-24T19:13:17.467307Z","shell.execute_reply":"2023-06-24T19:13:17.474683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reading the dataframe and getting the length ##","metadata":{}},{"cell_type":"code","source":"provided_dataframe = pd.read_csv(\"/kaggle/input/asl-fingerspelling/train.csv\")\ndisplay(provided_dataframe.head())","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:13:33.970685Z","iopub.execute_input":"2023-06-24T19:13:33.971002Z","iopub.status.idle":"2023-06-24T19:13:34.036499Z","shell.execute_reply.started":"2023-06-24T19:13:33.970981Z","shell.execute_reply":"2023-06-24T19:13:34.035585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(provided_dataframe)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:11:12.885086Z","iopub.execute_input":"2023-06-24T19:11:12.885408Z","iopub.status.idle":"2023-06-24T19:11:12.89136Z","shell.execute_reply.started":"2023-06-24T19:11:12.885385Z","shell.execute_reply":"2023-06-24T19:11:12.890473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reading a sample Parquest file ##","metadata":{}},{"cell_type":"code","source":"# tying out a aprquet file\n# Specify the path to the Parquet file\nparquet_file_path = '/kaggle/input/asl-fingerspelling/train_landmarks/5414471.parquet'\nsign = pd.read_parquet(parquet_file_path)   \nsequence = sign[sign.index == 1816796431]\ndisplay(sequence)\nprint(\"Length of this individual sequence is {}\".format(len(sequence)))\n","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:14:36.473173Z","iopub.execute_input":"2023-06-24T19:14:36.473496Z","iopub.status.idle":"2023-06-24T19:14:37.8198Z","shell.execute_reply.started":"2023-06-24T19:14:36.473473Z","shell.execute_reply":"2023-06-24T19:14:37.818936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing one of the sequences #","metadata":{}},{"cell_type":"code","source":"# Source :https://www.kaggle.com/code/omarsobhy14/asl-understanding-dataset\ndef map_new_to_old_style(sequence):\n    types = []\n    landmark_indexes = []\n    for column in list(sequence.columns)[1:544]:\n        parts = column.split(\"_\")\n        if len(parts) == 4:\n            types.append(parts[1] + \"_\" + parts[2])\n        else:\n            types.append(parts[1])\n\n        landmark_indexes.append(int(parts[-1]))\n\n    data = {\n        \"frame\": [],\n        \"type\": [],\n        \"landmark_index\": [],\n        \"x\": [],\n        \"y\": [],\n        \"z\": []\n    }\n\n    for index, row in sequence.iterrows():\n        data[\"frame\"] += [int(row.frame)]*543\n        data[\"type\"] += types\n        data[\"landmark_index\"] += landmark_indexes\n\n        for _type, landmark_index in zip(types, landmark_indexes):\n            data[\"x\"].append(row[f\"x_{_type}_{landmark_index}\"])\n            data[\"y\"].append(row[f\"y_{_type}_{landmark_index}\"])\n            data[\"z\"].append(row[f\"z_{_type}_{landmark_index}\"])\n\n    return pd.DataFrame.from_dict(data)\n\n# assign desired colors to landmarks\ndef assign_color(row):\n    if row == 'face':\n        return 'red'\n    elif 'hand' in row:\n        return 'dodgerblue'\n    else:\n        return 'green'\n\n# specifies the plotting order\ndef assign_order(row):\n    if row.type == 'face':\n        return row.landmark_index + 101\n    elif row.type == 'pose':\n        return row.landmark_index + 30\n    elif row.type == 'left_hand':\n        return row.landmark_index + 80\n    else:\n        return row.landmark_index\n    \ndef visualise2d_landmarks(parquet_df, title=\"\"):\n    connections = [  \n        [0, 1, 2, 3, 4,],\n        [0, 5, 6, 7, 8],\n        [0, 9, 10, 11, 12],\n        [0, 13, 14, 15, 16],\n        [0, 17, 18, 19, 20],\n\n        \n        [38, 36, 35, 34, 30, 31, 32, 33, 37],\n        [40, 39],\n        [52, 46, 50, 48, 46, 44, 42, 41, 43, 45, 47, 49, 45, 51],\n        [42, 54, 56, 58, 60, 62, 58],\n        [41, 53, 55, 57, 59, 61, 57],\n        [54, 53],\n\n        \n        [80, 81, 82, 83, 84, ],\n        [80, 85, 86, 87, 88],\n        [80, 89, 90, 91, 92],\n        [80, 93, 94, 95, 96],\n        [80, 97, 98, 99, 100], ]\n\n    parquet_df = map_new_to_old_style(parquet_df)\n    frames = sorted(set(parquet_df.frame))\n    first_frame = min(frames)\n    parquet_df['color'] = parquet_df.type.apply(lambda row: assign_color(row))\n    parquet_df['plot_order'] = parquet_df.apply(lambda row: assign_order(row), axis=1)\n    first_frame_df = parquet_df[parquet_df.frame == first_frame].copy()\n    first_frame_df = first_frame_df.sort_values([\"plot_order\"]).set_index('plot_order')\n\n\n    frames_l = []\n    for frame in frames:\n        filtered_df = parquet_df[parquet_df.frame == frame].copy()\n        filtered_df = filtered_df.sort_values([\"plot_order\"]).set_index(\"plot_order\")\n        traces = [go.Scatter(\n            x=filtered_df['x'],\n            y=filtered_df['y'],\n            mode='markers',\n            marker=dict(\n                color=filtered_df.color,\n                size=9))]\n\n        for i, seg in enumerate(connections):\n            trace = go.Scatter(\n                    x=filtered_df.loc[seg]['x'],\n                    y=filtered_df.loc[seg]['y'],\n                    mode='lines',\n            )\n            traces.append(trace)\n        frame_data = go.Frame(data=traces, traces = [i for i in range(17)])\n        frames_l.append(frame_data)\n\n    traces = [go.Scatter(\n        x=first_frame_df['x'],\n        y=first_frame_df['y'],\n        mode='markers',\n        marker=dict(\n            color=first_frame_df.color,\n            size=9\n        )\n    )]\n    for i, seg in enumerate(connections):\n        trace = go.Scatter(\n            x=first_frame_df.loc[seg]['x'],\n            y=first_frame_df.loc[seg]['y'],\n            mode='lines',\n            line=dict(\n                color='black',\n                width=2\n            )\n        )\n        traces.append(trace)\n    fig = go.Figure(\n        data=traces,\n        frames=frames_l\n    )\n\n\n    fig.update_layout(\n        width=800,\n        height=800,\n        scene={\n            'aspectmode': 'data',\n        },\n        updatemenus=[\n            {\n                \"buttons\": [\n                    {\n                        \"args\": [None, {\"frame\": {\"duration\": 100,\n                                                  \"redraw\": True},\n                                        \"fromcurrent\": True,\n                                        \"transition\": {\"duration\": 0}}],\n                        \"label\": \"&#9654;\",\n                        \"method\": \"animate\",\n                    },\n                    {\n                        \"args\": [[None], {\"frame\": {\"duration\": 0, \"redraw\": False},\n                                          \"mode\": \"immediate\",\n                                          \"transition\": {\"duration\": 0}}],\n                        \"label\": \"&#9612;&#9612;\",\n                        \"method\": \"animate\",\n                    },\n                ],\n                \"direction\": \"left\",\n                \"pad\": {\"r\": 100, \"t\": 100},\n                \"font\": {\"size\":20},\n                \"type\": \"buttons\",\n                \"x\": 0.1,\n                \"y\": 0,\n            }\n        ],\n    )\n    camera = dict(\n        up=dict(x=0, y=-1, z=0),\n        eye=dict(x=0, y=0, z=2.5)\n    )\n    fig.update_layout(title_text=title, title_x=0.5)\n    fig.update_layout(scene_camera=camera, showlegend=False)\n    fig.update_layout(xaxis = dict(visible=False),\n            yaxis = dict(visible=False),\n    )\n    fig.update_yaxes(autorange=\"reversed\")\n\n    fig.show()\n    \ndef get_phrase(df, file_id, sequence_id):\n    return df[\n        np.logical_and(\n            df.file_id == file_id, \n            df.sequence_id == sequence_id\n        )\n    ].phrase.iloc[0]\n\nimport pandas as pd,numpy as np,os\nimport json\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.io as pio\nfrom pathlib import Path\npio.templates.default = \"simple_white\"\nprint(\"importing..\")","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:14:42.741005Z","iopub.execute_input":"2023-06-24T19:14:42.741342Z","iopub.status.idle":"2023-06-24T19:14:44.929636Z","shell.execute_reply.started":"2023-06-24T19:14:42.74132Z","shell.execute_reply":"2023-06-24T19:14:44.928479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_id = 5414471\nsequence_id = 1816796431\nsequence_phrase = get_phrase(provided_dataframe, file_id, sequence_id)\nvisualise2d_landmarks(sequence, f\"Phrase: {sequence_phrase}\")\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:14:47.155584Z","iopub.execute_input":"2023-06-24T19:14:47.155897Z","iopub.status.idle":"2023-06-24T19:14:59.177796Z","shell.execute_reply.started":"2023-06-24T19:14:47.155872Z","shell.execute_reply":"2023-06-24T19:14:59.17708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# General training dataframe EDA #","metadata":{}},{"cell_type":"markdown","source":"# Character Length Occurances of Phrase #","metadata":{}},{"cell_type":"code","source":"def count_character_length(phrase):\n    return len(phrase)\n\nprovided_dataframe[\"pharse_length\"] = provided_dataframe[\"phrase\"].apply(count_character_length)\ndisplay(provided_dataframe.head())","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:15:20.205104Z","iopub.execute_input":"2023-06-24T19:15:20.205447Z","iopub.status.idle":"2023-06-24T19:15:20.241828Z","shell.execute_reply.started":"2023-06-24T19:15:20.205422Z","shell.execute_reply":"2023-06-24T19:15:20.240727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Maximum phrase length : {}\".format(max(provided_dataframe[\"pharse_length\"])))","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:15:26.728262Z","iopub.execute_input":"2023-06-24T19:15:26.728627Z","iopub.status.idle":"2023-06-24T19:15:26.740768Z","shell.execute_reply.started":"2023-06-24T19:15:26.728602Z","shell.execute_reply":"2023-06-24T19:15:26.739715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get the number of Unique Frame ids and file_ids #","metadata":{}},{"cell_type":"code","source":"unique_file_ids = provided_dataframe[\"file_id\"].unique()\ndisplay(unique_file_ids)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:15:50.356519Z","iopub.execute_input":"2023-06-24T19:15:50.356886Z","iopub.status.idle":"2023-06-24T19:15:50.363796Z","shell.execute_reply.started":"2023-06-24T19:15:50.356864Z","shell.execute_reply":"2023-06-24T19:15:50.36294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Gives us the Parquest files along with the number of sequences it contains\nfor i in range(len(unique_file_ids)):\n    temp_df = provided_dataframe[provided_dataframe['file_id'] == unique_file_ids[i]]\n    print(\"Unique_file_id : {}, Sequences : {}\".format(unique_file_ids[i], len(temp_df['sequence_id'].unique())))\n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:15:58.723343Z","iopub.execute_input":"2023-06-24T19:15:58.723679Z","iopub.status.idle":"2023-06-24T19:15:58.764562Z","shell.execute_reply.started":"2023-06-24T19:15:58.723655Z","shell.execute_reply":"2023-06-24T19:15:58.763405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Studying One Parquet File #","metadata":{}},{"cell_type":"code","source":"# loading a parquet file\n# tying out a aprquet file\n# Specify the path to the Parquet file\nparquet_file_path = '/kaggle/input/asl-fingerspelling/train_landmarks/5414471.parquet'\n\n# Read the Parquet file\ntable = pd.read_parquet(parquet_file_path)   \ndisplay(table)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:16:47.278284Z","iopub.execute_input":"2023-06-24T19:16:47.278601Z","iopub.status.idle":"2023-06-24T19:16:48.585097Z","shell.execute_reply.started":"2023-06-24T19:16:47.278577Z","shell.execute_reply":"2023-06-24T19:16:48.584283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# numbe of unique frames\nlen(table['frame'].unique())","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:16:52.387228Z","iopub.execute_input":"2023-06-24T19:16:52.387711Z","iopub.status.idle":"2023-06-24T19:16:52.394569Z","shell.execute_reply.started":"2023-06-24T19:16:52.387686Z","shell.execute_reply":"2023-06-24T19:16:52.39359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get the column names for sequence\nprint(\"Column Names : \")\nprint(table.columns)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:16:55.204829Z","iopub.execute_input":"2023-06-24T19:16:55.205172Z","iopub.status.idle":"2023-06-24T19:16:55.210256Z","shell.execute_reply.started":"2023-06-24T19:16:55.205148Z","shell.execute_reply":"2023-06-24T19:16:55.209432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating a sample dataframe that could be used for training ##\n### In this we sample randomly 25 parquet files and create a final dataframe that has some new columns that could be used for training ##","metadata":{}},{"cell_type":"code","source":"# get the list of columns with hand in it.\n#hand_cols = [col for col in table.columns if 'x_hand' or 'y_hand' in col]\nimport datetime\nhand_cols = []\nfor col in table.columns:\n    if('hand') in col:\n        if('x' or 'y') in col:\n            hand_cols.append(col)\nprint(hand_cols)\nprint(len(hand_cols))\n\ndef get_NAN_statistics(parquet_file_path, sequence_id):\n    # load the file\n    table = pd.read_parquet(parquet_file_path)\n    sequence = table[table.index == sequence_id]\n    nan_check = sequence[hand_cols].isna().any(axis=1)\n    num_rows_with_nan = nan_check.sum()\n    return num_rows_with_nan\n        \n        \n# Lets make  big dataframe that has all the information in one go\n\nfinal_df = provided_dataframe.sample(25, random_state=42)\nparquet_file_path_list = []\nnum_rows_with_nan = []\n\nfor i in range(len(final_df)):\n    parquet_file_name = final_df[\"file_id\"].iloc[i]\n    print(parquet_file_name)\n    # creating the file path\n    parquet_file_path = os.path.join(\"/kaggle/input/asl-fingerspelling/train_landmarks\", str(parquet_file_name) + \".parquet\")\n    parquet_file_path_list.append(os.path.join(\"/kaggle/input/asl-fingerspelling/train_landmarks\", str(parquet_file_name) + \".parquet\"))\n    # get the sequence id\n    sequence_id = final_df[\"sequence_id\"].iloc[i]\n    # get the number of nan rows\n    num_rows_with_nan.append(get_NAN_statistics(parquet_file_path, sequence_id))\n    \n    \nfinal_df[\"parquest_file_path\"] = parquet_file_path_list\nfinal_df[\"rows_with_nan\"] = num_rows_with_nan\n\nprint(final_df.head())\n    \n","metadata":{"execution":{"iopub.status.busy":"2023-06-23T17:42:08.285285Z","iopub.execute_input":"2023-06-23T17:42:08.28575Z","iopub.status.idle":"2023-06-23T17:44:32.427137Z","shell.execute_reply.started":"2023-06-23T17:42:08.285717Z","shell.execute_reply":"2023-06-23T17:44:32.425864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Assuming you have a DataFrame called 'df' and a list of columns called 'columns_list'\n# Check if there are any NaN values in the specified columns\nnan_check = df[columns_list].isna().any(axis=1)\n\n# Count the number of rows with NaN values\nnum_rows_with_nan = nan_check.sum()\n\n# Print the result\nprint(num_rows_with_nan)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Getting a statistics of the whole dataset #","metadata":{}},{"cell_type":"markdown","source":"## Video Statistics ##","metadata":{}},{"cell_type":"code","source":"# include filepath in the dataframe\ndef include_filepath(name):\n    return os.path.join(\"/kaggle/input/asl-fingerspelling/train_landmarks\", str(name) + \".parquet\")\n\nprovided_dataframe['file_path'] = provided_dataframe['file_id'].apply(include_filepath)\ndisplay(provided_dataframe.head())","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:17:50.367362Z","iopub.execute_input":"2023-06-24T19:17:50.367679Z","iopub.status.idle":"2023-06-24T19:17:50.479629Z","shell.execute_reply.started":"2023-06-24T19:17:50.367656Z","shell.execute_reply":"2023-06-24T19:17:50.478919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of Unique Frames in Recording\nfrom tqdm import tqdm\nN = 25 # Randomly \nN_UNIQUE_FRAMES = []\nUNIQUE_FILE_PATHS = pd.Series(provided_dataframe['file_path'].unique())\nSEED = 42\n\nfor idx, file_path in enumerate(tqdm(UNIQUE_FILE_PATHS.sample(N, random_state=SEED))):\n    df = pd.read_parquet(file_path)\n    for group, group_df in df.groupby('sequence_id'):\n        N_UNIQUE_FRAMES.append(group_df['frame'].nunique())\n\n# Convert to Numpy Array\nN_UNIQUE_FRAMES = np.array(N_UNIQUE_FRAMES)\nprint(N_UNIQUE_FRAMES)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:18:02.073694Z","iopub.execute_input":"2023-06-24T19:18:02.073994Z","iopub.status.idle":"2023-06-24T19:23:17.928739Z","shell.execute_reply.started":"2023-06-24T19:18:02.073971Z","shell.execute_reply":"2023-06-24T19:23:17.926905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot the frequencey for the number of unique frames in these 25 parquets files\nimport matplotlib.pyplot as plt\nplt.figure(figsize=(15,8))\nimport math\nplt.title('Number of Unique Frames', size=24)\npd.Series(N_UNIQUE_FRAMES).plot(kind='hist', bins=128)\nplt.grid()\nxlim = math.ceil(plt.xlim()[1])\nplt.xlim(0, xlim)\nplt.xticks(np.arange(0, xlim+50, 50))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:30:49.005501Z","iopub.execute_input":"2023-06-24T19:30:49.006192Z","iopub.status.idle":"2023-06-24T19:30:49.552221Z","shell.execute_reply.started":"2023-06-24T19:30:49.006134Z","shell.execute_reply":"2023-06-24T19:30:49.551077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Same way let's get the frequecy of non NAN frames ###","metadata":{}},{"cell_type":"code","source":"# rad one parquet file and determine the columns that we need\ndef determine_columns(filepath):\n    table = pd.read_parquet(filepath)\n    hand_cols = []\n    for col in table.columns:\n        if('hand') in col:\n            if('x' or 'y') in col:\n                hand_cols.append(col)\n    print(len(hand_cols))\n    return hand_cols\n\nhand_cols = determine_columns(\"/kaggle/input/asl-fingerspelling/train_landmarks/1019715464.parquet\")\nprint(\"Hand Columns : \")\nprint(hand_cols)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:30:52.904878Z","iopub.execute_input":"2023-06-24T19:30:52.905257Z","iopub.status.idle":"2023-06-24T19:30:55.564171Z","shell.execute_reply.started":"2023-06-24T19:30:52.905228Z","shell.execute_reply":"2023-06-24T19:30:55.562789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NAN_FRAMES = []\nTOTAL_FRAMES = []\nfor idx, file_path in enumerate(tqdm(UNIQUE_FILE_PATHS.sample(N, random_state=SEED))):\n    df = pd.read_parquet(file_path)\n    for group, group_df in df.groupby('sequence_id'):\n        TOTAL_FRAMES.append(group_df['frame'].nunique())\n        nan_check = group_df[hand_cols].isna().any(axis=1)\n        NAN_FRAMES.append(nan_check.sum())\n        \n        \n        ","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:31:01.235245Z","iopub.execute_input":"2023-06-24T19:31:01.235604Z","iopub.status.idle":"2023-06-24T19:36:53.248345Z","shell.execute_reply.started":"2023-06-24T19:31:01.235577Z","shell.execute_reply":"2023-06-24T19:36:53.247417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# plot the frequencey for the number of unique frames in these 25 parquets files\nimport matplotlib.pyplot as plt\nplt.figure(figsize=(15,8))\nimport math\nplt.title('Number of NAN Frames', size=24)\npd.Series(NAN_FRAMES).plot(kind='hist', bins=128, label = \"NAN\")\npd.Series(TOTAL_FRAMES).plot(kind='hist', bins=128, label = \"TOTAL\")\nplt.grid()\nxlim = math.ceil(plt.xlim()[1])\nplt.xlim(0, xlim)\nplt.xticks(np.arange(0, xlim+50, 50))\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-24T19:39:49.658728Z","iopub.execute_input":"2023-06-24T19:39:49.660265Z","iopub.status.idle":"2023-06-24T19:39:50.389791Z","shell.execute_reply.started":"2023-06-24T19:39:49.660209Z","shell.execute_reply":"2023-06-24T19:39:50.388655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Hence from the above graph we can see that all frames have NAN values ###","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}