{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"#import relevant libraries\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n#used in Python for working with Parquet files\nimport pyarrow.parquet as pq\n#for visualizations\nimport matplotlib.pyplot as plt\nimport plotly.graph_objects as go","metadata":{"execution":{"iopub.status.busy":"2023-07-27T10:07:21.168611Z","iopub.execute_input":"2023-07-27T10:07:21.169012Z","iopub.status.idle":"2023-07-27T10:07:21.174924Z","shell.execute_reply.started":"2023-07-27T10:07:21.168983Z","shell.execute_reply":"2023-07-27T10:07:21.173386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Look at the general data set content\ntrain_file_path = \"/kaggle/input/asl-fingerspelling/train.csv\"\ndf_train = pd.read_csv(train_file_path)\ndf_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-07-27T09:19:35.132285Z","iopub.execute_input":"2023-07-27T09:19:35.132735Z","iopub.status.idle":"2023-07-27T09:19:35.347291Z","shell.execute_reply.started":"2023-07-27T09:19:35.132701Z","shell.execute_reply":"2023-07-27T09:19:35.346116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#length of the dataset\nlen(df_train)","metadata":{"execution":{"iopub.status.busy":"2023-07-27T03:44:10.100521Z","iopub.execute_input":"2023-07-27T03:44:10.101573Z","iopub.status.idle":"2023-07-27T03:44:10.109237Z","shell.execute_reply.started":"2023-07-27T03:44:10.101526Z","shell.execute_reply":"2023-07-27T03:44:10.108065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#getting unique sequence_id,participant_id,Phrases\nprint(\"Total Unique sequence_id: \",df_train['sequence_id'].nunique())\nprint(\"Total Unique participant_id: \",df_train['participant_id'].nunique())\nprint(\"Total Unique Phrases: \",df_train['phrase'].nunique())","metadata":{"execution":{"iopub.status.busy":"2023-07-27T09:19:54.165548Z","iopub.execute_input":"2023-07-27T09:19:54.166839Z","iopub.status.idle":"2023-07-27T09:19:54.196869Z","shell.execute_reply.started":"2023-07-27T09:19:54.166783Z","shell.execute_reply":"2023-07-27T09:19:54.195683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#value counts of phrases\ndf_train[\"phrase\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-07-27T03:44:33.87609Z","iopub.execute_input":"2023-07-27T03:44:33.876492Z","iopub.status.idle":"2023-07-27T03:44:33.912382Z","shell.execute_reply.started":"2023-07-27T03:44:33.876455Z","shell.execute_reply":"2023-07-27T03:44:33.911294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualizing top signs of uniques phrases\nfig, ax = plt.subplots(figsize=(8, 8))\ndf_train[\"phrase\"].value_counts().head(20).sort_values(ascending=False).plot(\n    kind=\"barh\", ax=ax, title=\"Top Signs in Training Dataset\"\n)\nax.set_xlabel(\"Number of Training Examples\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-27T03:44:43.986449Z","iopub.execute_input":"2023-07-27T03:44:43.986854Z","iopub.status.idle":"2023-07-27T03:44:44.423034Z","shell.execute_reply.started":"2023-07-27T03:44:43.98682Z","shell.execute_reply":"2023-07-27T03:44:44.422233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print min and max values of phrases\nprint(df_train[\"phrase\"].value_counts().min())\nprint(df_train[\"phrase\"].value_counts().max())","metadata":{"execution":{"iopub.status.busy":"2023-07-27T03:44:54.582312Z","iopub.execute_input":"2023-07-27T03:44:54.583475Z","iopub.status.idle":"2023-07-27T03:44:54.63811Z","shell.execute_reply.started":"2023-07-27T03:44:54.583425Z","shell.execute_reply":"2023-07-27T03:44:54.63711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Studying one parquet file with specific sequence_id","metadata":{}},{"cell_type":"code","source":"# Specify the path to the Parquet file\n#file_id = 5414471\n#Parquet is an open-source columnar storage format that is designed to be highly efficient for both reading and writing large datasets.\nparquet_file_path = '/kaggle/input/asl-fingerspelling/train_landmarks/5414471.parquet'\n\n# Read the Parquet file\nparaquet_df = pd.read_parquet(parquet_file_path)  \nprint(\"Landmark data in paraquet file :\\n \")\nparaquet_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-27T10:18:28.751326Z","iopub.execute_input":"2023-07-27T10:18:28.751804Z","iopub.status.idle":"2023-07-27T10:18:31.297629Z","shell.execute_reply.started":"2023-07-27T10:18:28.751772Z","shell.execute_reply":"2023-07-27T10:18:31.296423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of unique sequence Id\nprint(\"Number of unique sequence id: \",paraquet_df.index.nunique())\nprint(\"Number of unique frame: \",paraquet_df.frame.nunique())","metadata":{"execution":{"iopub.status.busy":"2023-07-27T09:57:46.953947Z","iopub.execute_input":"2023-07-27T09:57:46.954396Z","iopub.status.idle":"2023-07-27T09:57:46.964048Z","shell.execute_reply.started":"2023-07-27T09:57:46.954332Z","shell.execute_reply":"2023-07-27T09:57:46.962788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**landmark files contain approximately 1,000 sequences**","metadata":{}},{"cell_type":"code","source":"paraquet_df[['frame']].groupby('sequence_id').count()","metadata":{"execution":{"iopub.status.busy":"2023-07-27T09:59:47.826185Z","iopub.execute_input":"2023-07-27T09:59:47.826657Z","iopub.status.idle":"2023-07-27T09:59:47.848293Z","shell.execute_reply.started":"2023-07-27T09:59:47.826619Z","shell.execute_reply":"2023-07-27T09:59:47.847088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paraquet_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-07-27T10:00:28.433441Z","iopub.execute_input":"2023-07-27T10:00:28.433941Z","iopub.status.idle":"2023-07-27T10:00:28.778333Z","shell.execute_reply.started":"2023-07-27T10:00:28.433901Z","shell.execute_reply":"2023-07-27T10:00:28.777223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#It checks if the \"sequence_id\" column in the DataFrame is equal to the value 1816796431.\n#retrieve the value of the \"phrase\" column for the first row that satisfies the condition\nphrase_string = df_train.loc[df_train[\"sequence_id\"] == 1816796431].phrase.values[0]","metadata":{"execution":{"iopub.status.busy":"2023-07-27T10:18:50.467465Z","iopub.execute_input":"2023-07-27T10:18:50.467872Z","iopub.status.idle":"2023-07-27T10:18:50.47478Z","shell.execute_reply.started":"2023-07-27T10:18:50.467842Z","shell.execute_reply":"2023-07-27T10:18:50.473619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#retrieve the rows with index label 1816796431\ntarget_phrase = paraquet_df.loc[1816796431]\ntarget_phrase","metadata":{"execution":{"iopub.status.busy":"2023-07-27T10:18:53.160618Z","iopub.execute_input":"2023-07-27T10:18:53.161014Z","iopub.status.idle":"2023-07-27T10:18:53.199714Z","shell.execute_reply.started":"2023-07-27T10:18:53.160985Z","shell.execute_reply":"2023-07-27T10:18:53.198552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def wide_to_long_format(sequence):\n    # extracting types and landmark_indexes from col names\n    types = []\n    landmark_indexes = []\n    for column in list(sequence.columns)[1:544]:\n        parts = column.split(\"_\")\n        # 'face', 'left_hand', 'pose', 'right_hand'\n        if len(parts) == 4:\n            types.append(parts[1] + \"_\" + parts[2])\n        else:\n            types.append(parts[1])\n\n        landmark_indexes.append(int(parts[-1]))\n\n\n    data = {\n        \"frame\": [],\n        \"type\": [],\n        \"landmark_index\": [],\n        \"x\": [],\n        \"y\": [],\n        \"z\": []\n    }\n\n    for index, row in sequence.iterrows():\n        data[\"frame\"] += [int(row.frame)]*543\n        data[\"type\"] += types\n        data[\"landmark_index\"] += landmark_indexes\n\n        for _type, landmark_index in zip(types, landmark_indexes):\n            data[\"x\"].append(row[f\"x_{_type}_{landmark_index}\"])\n            data[\"y\"].append(row[f\"y_{_type}_{landmark_index}\"])\n            data[\"z\"].append(row[f\"z_{_type}_{landmark_index}\"])\n\n    return pd.DataFrame.from_dict(data)\n\n# assign desired colors to landmarks\ndef assign_color(row):\n    if row == 'face':\n        return 'lightblue'\n    elif 'hand' in row:\n        return 'red'\n    else:\n        return 'purple'\n\n# specifies the plotting order \ndef assign_order(row):\n    if row.type == 'face':\n        return row.landmark_index + 101\n    elif row.type == 'pose':\n        return row.landmark_index + 30\n    elif row.type == 'left_hand':\n        return row.landmark_index + 80\n    else:\n        return row.landmark_index\n\n\ndef visualize2d_landmarks(df_parquet, title=\"\"):\n    connections = [  \n        [0, 1, 2, 3, 4,],\n        [0, 5, 6, 7, 8],\n        [0, 9, 10, 11, 12],\n        [0, 13, 14, 15, 16],\n        [0, 17, 18, 19, 20],\n\n        \n        [38, 36, 35, 34, 30, 31, 32, 33, 37],\n        [40, 39],\n        [52, 46, 50, 48, 46, 44, 42, 41, 43, 45, 47, 49, 45, 51],\n        [42, 54, 56, 58, 60, 62, 58],\n        [41, 53, 55, 57, 59, 61, 57],\n        [54, 53],\n\n        \n        [80, 81, 82, 83, 84, ],\n        [80, 85, 86, 87, 88],\n        [80, 89, 90, 91, 92],\n        [80, 93, 94, 95, 96],\n        [80, 97, 98, 99, 100], ]\n    \n    df_parquet = wide_to_long_format(df_parquet)\n    frames = sorted(set(df_parquet.frame))\n    first_frame = min(frames)\n    df_parquet['color'] = df_parquet.type.apply(lambda row: assign_color(row))\n    df_parquet['plot_order'] = df_parquet.apply(lambda row: assign_order(row), axis=1)\n    first_frame_df = df_parquet[df_parquet.frame == first_frame].copy()\n    first_frame_df = first_frame_df.sort_values([\"plot_order\"]).set_index('plot_order')\n\n\n    frames_l = []\n    for frame in frames:\n        filtered_df = df_parquet[df_parquet.frame == frame].copy()\n        filtered_df = filtered_df.sort_values([\"plot_order\"]).set_index(\"plot_order\")\n        \n        #scatter plot\n        traces = [go.Scatter(\n            x=filtered_df['x'],\n            y=filtered_df['y'],\n            mode='markers',\n            marker=dict(\n                color=filtered_df.color,\n                size=9))]\n\n        # drawing lines \n        for i, seg in enumerate(connections):\n            trace = go.Scatter(\n                    x=filtered_df.loc[seg]['x'],\n                    y=filtered_df.loc[seg]['y'],\n                    mode='lines',\n            )\n            traces.append(trace)\n        frame_data = go.Frame(data=traces, traces = [i for i in range(17)])\n        frames_l.append(frame_data)\n\n    # First frame -  scatter plot\n    traces = [go.Scatter(\n        x=first_frame_df['x'],\n        y=first_frame_df['y'],\n        mode='markers',\n        marker=dict(\n            color=first_frame_df.color,\n            size=9\n        )\n    )]\n    # Drawing lines in First Frame\n    for i, seg in enumerate(connections):\n        trace = go.Scatter(\n            x=first_frame_df.loc[seg]['x'],\n            y=first_frame_df.loc[seg]['y'],\n            mode='lines',\n            line=dict(\n                color='black',\n                width=2\n            )\n        )\n        traces.append(trace)\n    \n    fig = go.Figure(\n        data=traces,\n        frames=frames_l\n    )\n\n    fig.update_layout(\n        width=500,\n        height=800,\n        scene={\n            'aspectmode': 'data',\n        },\n        title_text=title,\n        title_x=0.5,\n        updatemenus=[\n            {\n                \"buttons\": [\n                    {\n                        \"args\": [None, {\"frame\": {\"duration\": 100,\n                                                  \"redraw\": True},\n                                        \"fromcurrent\": True,\n                                        \"transition\": {\"duration\": 0}}],\n                        \"label\": \"&#9654;\",\n                        \"method\": \"animate\",\n                    },\n\n                ],\n                \"direction\": \"left\",\n                \"pad\": {\"r\": 100, \"t\": 100},\n                \"font\": {\"size\":30},\n                \"type\": \"buttons\",\n                \"x\": 0.1,\n                \"y\": 0,\n            }\n        ],\n    )\n    # set up camera position \n    camera = dict(\n        up=dict(x=0, y=-1, z=0),\n        eye=dict(x=0, y=0, z=0)\n    )\n    fig.update_layout(scene_camera=camera, showlegend=False)\n    fig.update_layout(xaxis = dict(visible=False),\n            yaxis = dict(visible=False),\n    )\n    fig.update_yaxes(autorange=\"reversed\")\n\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-27T10:30:18.536587Z","iopub.execute_input":"2023-07-27T10:30:18.53702Z","iopub.status.idle":"2023-07-27T10:30:18.57179Z","shell.execute_reply.started":"2023-07-27T10:30:18.536978Z","shell.execute_reply":"2023-07-27T10:30:18.570696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#retrieving data from paraquet_df when sequence_id = 1816796431 \nsequence_id = 1816796431 #parq_example_df.index[0]\nseq_df = paraquet_df[paraquet_df.index==sequence_id]\nseq_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-27T10:19:26.78084Z","iopub.execute_input":"2023-07-27T10:19:26.781269Z","iopub.status.idle":"2023-07-27T10:19:26.790813Z","shell.execute_reply.started":"2023-07-27T10:19:26.781238Z","shell.execute_reply":"2023-07-27T10:19:26.789999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# number of unique frames\nlen(sequence['frame'].unique())","metadata":{"execution":{"iopub.status.busy":"2023-07-27T10:19:29.537399Z","iopub.execute_input":"2023-07-27T10:19:29.537825Z","iopub.status.idle":"2023-07-27T10:19:29.545549Z","shell.execute_reply.started":"2023-07-27T10:19:29.537793Z","shell.execute_reply":"2023-07-27T10:19:29.544377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualize the landmark\nvisualize2d_landmarks(seq_df, phrase_string)","metadata":{"execution":{"iopub.status.busy":"2023-07-27T10:19:32.032345Z","iopub.execute_input":"2023-07-27T10:19:32.032762Z","iopub.status.idle":"2023-07-27T10:19:51.657583Z","shell.execute_reply.started":"2023-07-27T10:19:32.03273Z","shell.execute_reply":"2023-07-27T10:19:51.656639Z"},"trusted":true},"execution_count":null,"outputs":[]}]}