{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:22.060501Z","iopub.execute_input":"2023-05-29T10:36:22.06165Z","iopub.status.idle":"2023-05-29T10:36:22.108961Z","shell.execute_reply.started":"2023-05-29T10:36:22.06158Z","shell.execute_reply":"2023-05-29T10:36:22.107547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# helper functions\ndef map_new_to_old_style(sequence):\n    # there are 4 types of landmarks [face,pose,right_hand,left_hand]\n    types = []# list where we'll store landmarks\n    landmark_indexes = []\n    for column in list(sequence.columns)[1:544]:\n        # first column is frame which is not a landmark so counter starts from 1\n        # why 544 - because we are given that there are now 1,629 spatial coordinate columns for the x, y and z coordinates for each of the 543 landmarks.\n        parts = column.split(\"_\")\n        if len(parts) == 4:\n            # for x_left_hand_1 - there will be 4 parts and for x_pose_1 there will be 3 only\n            types.append(parts[1] + \"_\" + parts[2])\n        else:\n            types.append(parts[1])\n\n        landmark_indexes.append(int(parts[-1]))\n\n    data = {\"frame\": [], \"type\": [], \"landmark_index\": [], \"x\": [], \"y\": [], \"z\": []}\n\n    for index, row in sequence.iterrows():\n        data[\"frame\"] += [int(row.frame)] * 543\n        #       frame has values from 1,2,3....\n        # 1. Firstly we are converting them into integers - 1\n        # 2. Then we are making it a list - [1]\n        # 3. Multiplying by 543 makes the replicates the same element(1 from our example) 543 times for all the landmarks present\n        data[\"type\"] += types\n        data[\"landmark_index\"] += landmark_indexes\n\n        for _type, landmark_index in zip(types, landmark_indexes):\n            data[\"x\"].append(row[f\"x_{_type}_{landmark_index}\"])\n            data[\"y\"].append(row[f\"y_{_type}_{landmark_index}\"])\n            data[\"z\"].append(row[f\"z_{_type}_{landmark_index}\"])\n\n    return pd.DataFrame.from_dict(data)\n\n\ndef assign_colors(row):\n    if row == \"face\":\n        return \"red\"\n    elif \"hand\" in row:\n        return \"blue\"\n    else:\n        return \"green\"\n\n\n# specifies the plotting order\ndef assign_order(row):\n    if row.type == \"face\":\n        return row.landmark_index + 101\n    elif row.type == \"pose\":\n        return row.landmark_index + 30\n    elif row.type == \"left_hand\":\n        return row.landmark_index + 80\n    else:\n        return row.landmark_index","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:22.113782Z","iopub.execute_input":"2023-05-29T10:36:22.1147Z","iopub.status.idle":"2023-05-29T10:36:22.127824Z","shell.execute_reply.started":"2023-05-29T10:36:22.114663Z","shell.execute_reply":"2023-05-29T10:36:22.126833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# A function to visualize the landmarks in 2d\ndef visualise2d_landmarks(parquet_df, title=\"\"):\n    # we first define a list of landmark connections, which specify which landmarks are connected by a line.\n\n    # face landmarks are not connected by lines, you can also see that we have added 101 to face landmark indexs and in connections all values are below 100\n\n    connections = [  \n        [0, 1, 2, 3, 4,],\n        [0, 5, 6, 7, 8],\n        [0, 9, 10, 11, 12],\n        [0, 13, 14, 15, 16],\n        [0, 17, 18, 19, 20],\n\n        \n        [38, 36, 35, 34, 30, 31, 32, 33, 37],\n        [40, 39],\n        [52, 46, 50, 48, 46, 44, 42, 41, 43, 45, 47, 49, 45, 51],\n        [42, 54, 56, 58, 60, 62, 58],\n        [41, 53, 55, 57, 59, 61, 57],\n        [54, 53],\n\n        \n        [80, 81, 82, 83, 84, ],\n        [80, 85, 86, 87, 88],\n        [80, 89, 90, 91, 92],\n        [80, 93, 94, 95, 96],\n        [80, 97, 98, 99, 100], ]\n\n    parquet_df = map_new_to_old_style(parquet_df)\n    frames = sorted(set(parquet_df.frame))\n    first_frame = min(frames)\n    parquet_df[\"color\"] = parquet_df.type.apply(lambda row: assign_colors(row))\n    parquet_df[\"plot_order\"] = parquet_df.apply(lambda row: assign_order(row), axis=1)\n    first_frame_df = parquet_df[parquet_df.frame == first_frame].copy()\n    first_frame_df = first_frame_df.sort_values([\"plot_order\"]).set_index(\"plot_order\")\n\n    frames_l = []\n    for frame in frames:\n        filtered_df = parquet_df[parquet_df.frame == frame].copy()\n        filtered_df = filtered_df.sort_values([\"plot_order\"]).set_index(\"plot_order\")\n        traces = [\n            go.Scatter(\n                x=filtered_df[\"x\"],\n                y=filtered_df[\"y\"],\n                mode=\"markers\",\n                marker=dict(color=filtered_df.color, size=9),\n            )\n        ]\n\n        for i, seg in enumerate(connections):\n            trace = go.Scatter(\n                x=filtered_df.loc[seg][\"x\"],\n                y=filtered_df.loc[seg][\"y\"],\n                mode=\"lines\",\n            )\n            traces.append(trace)\n        frame_data = go.Frame(data=traces, traces=[i for i in range(17)])\n        frames_l.append(frame_data)\n\n    traces = [\n        go.Scatter(\n            x=first_frame_df[\"x\"],\n            y=first_frame_df[\"y\"],\n            mode=\"markers\",\n            marker=dict(color=first_frame_df.color, size=9),\n        )\n    ]\n    for i, seg in enumerate(connections):\n        trace = go.Scatter(\n            x=first_frame_df.loc[seg][\"x\"],\n            y=first_frame_df.loc[seg][\"y\"],\n            mode=\"lines\",\n            line=dict(color=\"black\", width=2),\n        )\n        traces.append(trace)\n    fig = go.Figure(data=traces, frames=frames_l)\n\n    fig.update_layout(\n        width=500,\n        height=800,\n        scene={\n            \"aspectmode\": \"data\",\n        },\n        updatemenus=[\n            {\n                \"buttons\": [\n                    {\n                        \"args\": [\n                            None,\n                            {\n                                \"frame\": {\"duration\": 100, \"redraw\": True},\n                                \"fromcurrent\": True,\n                                \"transition\": {\"duration\": 0},\n                            },\n                        ],\n                        \"label\": \"&#9654;\",\n                        \"method\": \"animate\",\n                    },\n                    {\n                        \"args\": [\n                            [None],\n                            {\n                                \"frame\": {\"duration\": 0, \"redraw\": False},\n                                \"mode\": \"immediate\",\n                                \"transition\": {\"duration\": 0},\n                            },\n                        ],\n                        \"label\": \"&#9612;&#9612;\",\n                        \"method\": \"animate\",\n                    },\n                ],\n                \"direction\": \"left\",\n                \"pad\": {\"r\": 100, \"t\": 100},\n                \"font\": {\"size\": 20},\n                \"type\": \"buttons\",\n                \"x\": 0.1,\n                \"y\": 0,\n            }\n        ],\n    )\n    camera = dict(up=dict(x=0, y=-1, z=0), eye=dict(x=0, y=0, z=2.5))\n    fig.update_layout(title_text=title, title_x=0.5)\n    fig.update_layout(scene_camera=camera, showlegend=False)\n    fig.update_layout(\n        xaxis=dict(visible=False),\n        yaxis=dict(visible=False),\n    )\n    fig.update_yaxes(autorange=\"reversed\")\n\n    fig.show()\n\n\ndef get_phrase(df, file_id, sequence_id):\n    return df[\n        np.logical_and(df.file_id == file_id, df.sequence_id == sequence_id)\n    ].phrase.iloc[0]","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:22.134373Z","iopub.execute_input":"2023-05-29T10:36:22.134739Z","iopub.status.idle":"2023-05-29T10:36:22.165359Z","shell.execute_reply.started":"2023-05-29T10:36:22.134708Z","shell.execute_reply":"2023-05-29T10:36:22.163952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# Importing Libraries\n\n\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport plotly.express as px\nimport plotly.graph_objs as go\nimport json\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:22.168111Z","iopub.execute_input":"2023-05-29T10:36:22.168675Z","iopub.status.idle":"2023-05-29T10:36:24.751367Z","shell.execute_reply.started":"2023-05-29T10:36:22.168628Z","shell.execute_reply":"2023-05-29T10:36:24.750095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"supplement_data = pd.read_csv('/kaggle/input/asl-fingerspelling/supplemental_metadata.csv')\ntrain_data = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:24.75358Z","iopub.execute_input":"2023-05-29T10:36:24.754012Z","iopub.status.idle":"2023-05-29T10:36:25.086401Z","shell.execute_reply.started":"2023-05-29T10:36:24.753972Z","shell.execute_reply":"2023-05-29T10:36:25.085305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:25.097374Z","iopub.execute_input":"2023-05-29T10:36:25.097951Z","iopub.status.idle":"2023-05-29T10:36:25.105671Z","shell.execute_reply.started":"2023-05-29T10:36:25.097906Z","shell.execute_reply":"2023-05-29T10:36:25.104424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"supplement_data.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:25.108238Z","iopub.execute_input":"2023-05-29T10:36:25.109556Z","iopub.status.idle":"2023-05-29T10:36:25.12395Z","shell.execute_reply.started":"2023-05-29T10:36:25.109509Z","shell.execute_reply":"2023-05-29T10:36:25.122422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{"execution":{"iopub.status.busy":"2023-05-27T14:39:12.418218Z","iopub.execute_input":"2023-05-27T14:39:12.418655Z","iopub.status.idle":"2023-05-27T14:39:12.424913Z","shell.execute_reply.started":"2023-05-27T14:39:12.418623Z","shell.execute_reply":"2023-05-27T14:39:12.423624Z"}}},{"cell_type":"code","source":"supplement_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:25.125717Z","iopub.execute_input":"2023-05-29T10:36:25.126299Z","iopub.status.idle":"2023-05-29T10:36:25.161967Z","shell.execute_reply.started":"2023-05-29T10:36:25.126263Z","shell.execute_reply":"2023-05-29T10:36:25.160782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of rows and columns:\", supplement_data.shape) \nprint()\nprint('*************************************')\nprint()\nprint(supplement_data.info())\nprint()\nprint('*************************************')\nprint()\nprint(\"Number of Unique Values:\\n\",supplement_data.nunique())\nprint()\nprint('*************************************')\nprint()\nprint(\"Duplicate Values:\\n\",supplement_data.duplicated().sum())\nprint()\nprint('*************************************')\nprint()\nprint(\"Null Values:\\n\",supplement_data.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:25.163636Z","iopub.execute_input":"2023-05-29T10:36:25.164642Z","iopub.status.idle":"2023-05-29T10:36:25.36828Z","shell.execute_reply.started":"2023-05-29T10:36:25.1646Z","shell.execute_reply":"2023-05-29T10:36:25.367237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\nObservations :\n\n    There are 52958 rows and 5 columns.\n    There are no null values.\n    There are no duplicate records.\n\n","metadata":{}},{"cell_type":"code","source":"supplement_data.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:25.372441Z","iopub.execute_input":"2023-05-29T10:36:25.373146Z","iopub.status.idle":"2023-05-29T10:36:25.40492Z","shell.execute_reply.started":"2023-05-29T10:36:25.373106Z","shell.execute_reply":"2023-05-29T10:36:25.403775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# For Train data\n","metadata":{}},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:25.406721Z","iopub.execute_input":"2023-05-29T10:36:25.407462Z","iopub.status.idle":"2023-05-29T10:36:25.422635Z","shell.execute_reply.started":"2023-05-29T10:36:25.40742Z","shell.execute_reply":"2023-05-29T10:36:25.421311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of rows and columns:\", train_data.shape) \nprint()\nprint('*************************************')\nprint()\nprint(train_data.info())\nprint()\nprint('*************************************')\nprint()\nprint(\"Number of Unique Values:\\n\",train_data.nunique())\nprint()\nprint('*************************************')\nprint()\nprint(\"Duplicate Values:\\n\",train_data.duplicated().sum())\nprint()\nprint('*************************************')\nprint()\nprint(\"Null Values:\\n\",train_data.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:25.424585Z","iopub.execute_input":"2023-05-29T10:36:25.425402Z","iopub.status.idle":"2023-05-29T10:36:25.599299Z","shell.execute_reply.started":"2023-05-29T10:36:25.425328Z","shell.execute_reply":"2023-05-29T10:36:25.598104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\nObservations :\n\n    There are 67287 rows and 5 columns.\n    There are no null values.\n    There are no duplicate records.\n\n","metadata":{}},{"cell_type":"markdown","source":"\n# Getting to know the length of each phrase.\n","metadata":{}},{"cell_type":"code","source":"supplement_data['phrase_char_len'] = supplement_data['phrase'].apply(len)\nsupplement_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:25.601156Z","iopub.execute_input":"2023-05-29T10:36:25.601926Z","iopub.status.idle":"2023-05-29T10:36:25.648269Z","shell.execute_reply.started":"2023-05-29T10:36:25.601881Z","shell.execute_reply":"2023-05-29T10:36:25.64704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(supplement_data,x='phrase_char_len',nbins=35,color_discrete_sequence = px.colors.qualitative.Set2, title=\"Virus affected different age groups\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:25.650276Z","iopub.execute_input":"2023-05-29T10:36:25.650757Z","iopub.status.idle":"2023-05-29T10:36:27.82418Z","shell.execute_reply.started":"2023-05-29T10:36:25.65071Z","shell.execute_reply":"2023-05-29T10:36:27.822855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\nObservations :\n\n    It seems that the number of records are having maximum count for the phrase length between 25-35.\n\n","metadata":{}},{"cell_type":"code","source":"total_phrases = supplement_data['phrase'].value_counts()\n\ndata_phrases = pd.DataFrame({'phrases': total_phrases.index, 'phrase count': total_phrases.values})","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:27.825872Z","iopub.execute_input":"2023-05-29T10:36:27.826273Z","iopub.status.idle":"2023-05-29T10:36:27.842114Z","shell.execute_reply.started":"2023-05-29T10:36:27.826237Z","shell.execute_reply":"2023-05-29T10:36:27.841197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_phrases.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:27.845186Z","iopub.execute_input":"2023-05-29T10:36:27.846422Z","iopub.status.idle":"2023-05-29T10:36:27.874303Z","shell.execute_reply.started":"2023-05-29T10:36:27.846359Z","shell.execute_reply":"2023-05-29T10:36:27.873226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_phrases.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:27.87601Z","iopub.execute_input":"2023-05-29T10:36:27.87737Z","iopub.status.idle":"2023-05-29T10:36:27.890382Z","shell.execute_reply.started":"2023-05-29T10:36:27.877314Z","shell.execute_reply":"2023-05-29T10:36:27.889241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing the phrases count in a 'phrase' column.","metadata":{}},{"cell_type":"code","source":"fig = px.bar(data_phrases.iloc[:10,:],x='phrase count', y='phrases', height=700, width= 1200, color='phrases',color_discrete_sequence =px.colors.qualitative.G10, title='Top 10 Most frequently used phrases')\nfig.update_layout(xaxis={'categoryorder':'total descending'})\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:27.891943Z","iopub.execute_input":"2023-05-29T10:36:27.892384Z","iopub.status.idle":"2023-05-29T10:36:28.243962Z","shell.execute_reply.started":"2023-05-29T10:36:27.892285Z","shell.execute_reply":"2023-05-29T10:36:28.242601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# Least used phrases.\n","metadata":{}},{"cell_type":"code","source":"fig = px.bar(data_phrases.iloc[-10:,:],x='phrase count', y='phrases', height=700, width= 1200, color='phrases',color_discrete_sequence= px.colors.sequential.Plasma, title='Top 10 Least used phrases')\nfig.update_layout(xaxis={'categoryorder':'total descending'})\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:36:28.245649Z","iopub.execute_input":"2023-05-29T10:36:28.246428Z","iopub.status.idle":"2023-05-29T10:36:28.388355Z","shell.execute_reply.started":"2023-05-29T10:36:28.246383Z","shell.execute_reply":"2023-05-29T10:36:28.387021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploring landmarks file\n# Most used phrase","metadata":{}},{"cell_type":"code","source":"# Creating subset of dataset where phrase is \"why do you ask silly questions\".\n\ntop_phrase = supplement_data[supplement_data[\"phrase\"]==\"why do you ask silly questions\"]['path'].values[0]\ntop_phrase","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:47:25.098667Z","iopub.execute_input":"2023-05-29T10:47:25.099193Z","iopub.status.idle":"2023-05-29T10:47:25.12441Z","shell.execute_reply.started":"2023-05-29T10:47:25.099153Z","shell.execute_reply":"2023-05-29T10:47:25.123067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Least used phrase","metadata":{}},{"cell_type":"code","source":"# Creating subset of dataset where phrase is \"my favorite place is to visit\".\n\nbottom_phrase = supplement_data[supplement_data[\"phrase\"]==\"my favorite place is to visit\"]['path'].values[0]\nbottom_phrase","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:48:17.529387Z","iopub.execute_input":"2023-05-29T10:48:17.529792Z","iopub.status.idle":"2023-05-29T10:48:17.546906Z","shell.execute_reply.started":"2023-05-29T10:48:17.52976Z","shell.execute_reply":"2023-05-29T10:48:17.545956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploring data related to top_phrase further","metadata":{}},{"cell_type":"code","source":"landmark_df = pd.read_parquet('/kaggle/input/asl-fingerspelling/'+ top_phrase)\nlandmark_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:49:03.408446Z","iopub.execute_input":"2023-05-29T10:49:03.408926Z","iopub.status.idle":"2023-05-29T10:49:23.377882Z","shell.execute_reply.started":"2023-05-29T10:49:03.408885Z","shell.execute_reply":"2023-05-29T10:49:23.376677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"landmark_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:49:37.372936Z","iopub.execute_input":"2023-05-29T10:49:37.373513Z","iopub.status.idle":"2023-05-29T10:49:37.382254Z","shell.execute_reply.started":"2023-05-29T10:49:37.373463Z","shell.execute_reply":"2023-05-29T10:49:37.381083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"landmark_df=landmark_df.reset_index(inplace=False)\nlandmark_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:49:46.116615Z","iopub.execute_input":"2023-05-29T10:49:46.117109Z","iopub.status.idle":"2023-05-29T10:49:47.602262Z","shell.execute_reply.started":"2023-05-29T10:49:46.117031Z","shell.execute_reply":"2023-05-29T10:49:47.601103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"landmark_df.shape, landmark_df[\"sequence_id\"].nunique()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:50:07.403351Z","iopub.execute_input":"2023-05-29T10:50:07.403776Z","iopub.status.idle":"2023-05-29T10:50:07.41464Z","shell.execute_reply.started":"2023-05-29T10:50:07.403743Z","shell.execute_reply":"2023-05-29T10:50:07.413385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n    There are 171404 rows and 1631 columns.\n    There are 1000 unique sequence_id values.\n","metadata":{}},{"cell_type":"markdown","source":"# Reading Json Data","metadata":{}},{"cell_type":"code","source":"file_path = '/kaggle/input/asl-fingerspelling/character_to_prediction_index.json'\n\n# Open the JSON file and load its contents\nwith open(file_path, 'r') as file:\n    data = json.load(file)\n\nprint(data)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:52:54.668022Z","iopub.execute_input":"2023-05-29T10:52:54.668531Z","iopub.status.idle":"2023-05-29T10:52:54.679833Z","shell.execute_reply.started":"2023-05-29T10:52:54.668496Z","shell.execute_reply":"2023-05-29T10:52:54.678319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\n# Observations :\n\n    It seems there is a relation assigned between character and value.\n    We need to put it into more simple format.\n\n","metadata":{}},{"cell_type":"code","source":"char=[]\nvalue=[]\n\nfor i,j in data.items():\n    char.append(i)\n    value.append(j)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:54:10.000873Z","iopub.execute_input":"2023-05-29T10:54:10.001414Z","iopub.status.idle":"2023-05-29T10:54:10.008854Z","shell.execute_reply.started":"2023-05-29T10:54:10.001366Z","shell.execute_reply":"2023-05-29T10:54:10.007456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"char_to_pred_index=pd.DataFrame({\"char\":char,\"value\":value})\nchar_to_pred_index.head(20)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:54:23.54724Z","iopub.execute_input":"2023-05-29T10:54:23.547675Z","iopub.status.idle":"2023-05-29T10:54:23.562776Z","shell.execute_reply.started":"2023-05-29T10:54:23.547642Z","shell.execute_reply":"2023-05-29T10:54:23.561386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"char_to_pred_index.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:54:39.426476Z","iopub.execute_input":"2023-05-29T10:54:39.426897Z","iopub.status.idle":"2023-05-29T10:54:39.43399Z","shell.execute_reply.started":"2023-05-29T10:54:39.426864Z","shell.execute_reply":"2023-05-29T10:54:39.432854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing the Sequence\n","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:55:06.171661Z","iopub.execute_input":"2023-05-29T10:55:06.172121Z","iopub.status.idle":"2023-05-29T10:55:06.177672Z","shell.execute_reply.started":"2023-05-29T10:55:06.172086Z","shell.execute_reply":"2023-05-29T10:55:06.176306Z"}}},{"cell_type":"markdown","source":"\n# Generating a random integer between 0 and 53 (unique_file_ids)\n","metadata":{}},{"cell_type":"code","source":"\nunique_file_ids = len(np.unique(supplement_data['file_id']))\nunique_file_ids","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:56:30.470831Z","iopub.execute_input":"2023-05-29T10:56:30.471251Z","iopub.status.idle":"2023-05-29T10:56:30.480429Z","shell.execute_reply.started":"2023-05-29T10:56:30.471218Z","shell.execute_reply":"2023-05-29T10:56:30.479358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_id = np.random.randint(0, unique_file_ids)\n\n# Getting the random file_id \nrandom_file_id = np.unique(supplement_data['file_id'])[random_id]","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:56:31.373521Z","iopub.execute_input":"2023-05-29T10:56:31.374709Z","iopub.status.idle":"2023-05-29T10:56:31.381062Z","shell.execute_reply.started":"2023-05-29T10:56:31.374668Z","shell.execute_reply":"2023-05-29T10:56:31.379843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Getting all different sequences in random file","metadata":{}},{"cell_type":"code","source":"signs = supplement_data[supplement_data['file_id'] == random_file_id]\nsigns.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:57:53.494789Z","iopub.execute_input":"2023-05-29T10:57:53.495298Z","iopub.status.idle":"2023-05-29T10:57:53.511803Z","shell.execute_reply.started":"2023-05-29T10:57:53.495258Z","shell.execute_reply":"2023-05-29T10:57:53.510865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"signs.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:58:00.497719Z","iopub.execute_input":"2023-05-29T10:58:00.498156Z","iopub.status.idle":"2023-05-29T10:58:00.505969Z","shell.execute_reply.started":"2023-05-29T10:58:00.49812Z","shell.execute_reply":"2023-05-29T10:58:00.504724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Unique indexes : ',len(np.unique(signs.index)))","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:58:22.488954Z","iopub.execute_input":"2023-05-29T10:58:22.489507Z","iopub.status.idle":"2023-05-29T10:58:22.496159Z","shell.execute_reply.started":"2023-05-29T10:58:22.489464Z","shell.execute_reply":"2023-05-29T10:58:22.494855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# Getting a random Sequence id\n","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:59:08.635727Z","iopub.execute_input":"2023-05-29T10:59:08.636186Z","iopub.status.idle":"2023-05-29T10:59:08.642283Z","shell.execute_reply.started":"2023-05-29T10:59:08.636153Z","shell.execute_reply":"2023-05-29T10:59:08.640507Z"}}},{"cell_type":"code","source":"random_squence_id = signs.sample()['sequence_id'].item()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:59:21.201608Z","iopub.execute_input":"2023-05-29T10:59:21.202047Z","iopub.status.idle":"2023-05-29T10:59:21.21012Z","shell.execute_reply.started":"2023-05-29T10:59:21.202013Z","shell.execute_reply":"2023-05-29T10:59:21.208636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# Locating the simultaneous parquet record using the random_file_id generated","metadata":{}},{"cell_type":"code","source":"path_to_sign = f\"/kaggle/input/asl-fingerspelling/supplemental_landmarks/{random_file_id}.parquet\"\nparquet = pd.read_parquet(path_to_sign)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:59:46.329101Z","iopub.execute_input":"2023-05-29T10:59:46.330015Z","iopub.status.idle":"2023-05-29T11:00:06.67418Z","shell.execute_reply.started":"2023-05-29T10:59:46.32997Z","shell.execute_reply":"2023-05-29T11:00:06.672915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequence = parquet[parquet.index == random_squence_id]\nsequence","metadata":{"execution":{"iopub.status.busy":"2023-05-29T11:00:36.12765Z","iopub.execute_input":"2023-05-29T11:00:36.128137Z","iopub.status.idle":"2023-05-29T11:00:36.166275Z","shell.execute_reply.started":"2023-05-29T11:00:36.128097Z","shell.execute_reply":"2023-05-29T11:00:36.16494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing the Fingerspelling Animation","metadata":{}},{"cell_type":"code","source":"sequence_phrase = get_phrase(supplement_data, random_file_id, random_squence_id)\nvisualise2d_landmarks(sequence, f\"Phrase: {sequence_phrase}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-29T11:01:25.611971Z","iopub.execute_input":"2023-05-29T11:01:25.612454Z","iopub.status.idle":"2023-05-29T11:01:51.722012Z","shell.execute_reply.started":"2023-05-29T11:01:25.612418Z","shell.execute_reply":"2023-05-29T11:01:51.720402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}