{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport json\nimport plotly.graph_objects as go\nimport plotly.io as pio\n","metadata":{"execution":{"iopub.status.busy":"2023-06-30T18:40:58.993787Z","iopub.execute_input":"2023-06-30T18:40:58.994215Z","iopub.status.idle":"2023-06-30T18:40:58.999133Z","shell.execute_reply.started":"2023-06-30T18:40:58.994182Z","shell.execute_reply":"2023-06-30T18:40:58.997955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-30T18:40:59.036955Z","iopub.execute_input":"2023-06-30T18:40:59.037373Z","iopub.status.idle":"2023-06-30T18:40:59.048982Z","shell.execute_reply.started":"2023-06-30T18:40:59.037339Z","shell.execute_reply":"2023-06-30T18:40:59.047863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The code consists of several functions for visualizing 2D landmarks and manipulating data. Here's a description of each function:\n\n1. `map_new_to_old_style(sequence)`: This function takes a sequence DataFrame as input and maps the column names to an older style. It extracts types and landmark indexes from the column names and constructs a new DataFrame with specific columns.\n\n2. `assign_color(row)`: This function assigns colors to different types of landmarks based on the input row. Landmarks of type 'face' are assigned the color 'red', 'hand' landmarks are assigned 'dodgerblue', and others are assigned 'green'.\n\n3. `assign_order(row)`: This function assigns a plotting order to each row based on the type of landmark. The order is determined by adding an offset value to the landmark index.\n\n4. `visualise2d_landmarks(parquet_df, title=\"\")`: This function visualizes 2D landmarks using the Plotly library. It takes a DataFrame `parquet_df` containing the landmark data and an optional `title` for the plot. The function creates scatter plots for each frame, connecting the landmarks with lines. It also includes buttons for animation and allows interaction with the plot.\n\n5. `get_phrase(df, file_id, sequence_id)`: This function retrieves a phrase by filtering the DataFrame `df` based on `file_id` and `sequence_id`. It returns the corresponding phrase from the 'phrase' column.\n\nThe code has been optimized for better performance and readability. It includes comments to explain the purpose and functionality of each function.\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport plotly.graph_objects as go\n\ndef map_new_to_old_style(sequence):\n    types = []\n    landmark_indexes = []\n\n    # Extracting types and landmark indexes from column names\n    for column in list(sequence.columns)[1:544]:\n        parts = column.split(\"_\")\n        if len(parts) == 4:\n            types.append(parts[1] + \"_\" + parts[2])\n        else:\n            types.append(parts[1])\n        landmark_indexes.append(int(parts[-1]))\n\n    data = {\n        \"frame\": [],\n        \"type\": [],\n        \"landmark_index\": [],\n        \"x\": [],\n        \"y\": [],\n        \"z\": []\n    }\n\n    # Constructing the data dictionary\n    for index, row in sequence.iterrows():\n        data[\"frame\"] += [int(row.frame)] * 543\n        data[\"type\"] += types\n        data[\"landmark_index\"] += landmark_indexes\n\n        for _type, landmark_index in zip(types, landmark_indexes):\n            data[\"x\"].append(row[f\"x_{_type}_{landmark_index}\"])\n            data[\"y\"].append(row[f\"y_{_type}_{landmark_index}\"])\n            data[\"z\"].append(row[f\"z_{_type}_{landmark_index}\"])\n\n    return pd.DataFrame.from_dict(data)\n\ndef assign_color(row):\n    # Assigning colors based on the type of landmark\n    if row == 'face':\n        return 'rgb(255, 0, 0)'  # Red\n    elif 'hand' in row:\n        return 'rgb(30, 144, 255)'  # Dodger Blue\n    else:\n        return 'rgb(0, 255, 0)'  # Green\n\ndef assign_order(row):\n    # Assigning plotting order based on the type of landmark\n    if row.type == 'face':\n        return row.landmark_index + 101\n    elif row.type == 'pose':\n        return row.landmark_index + 30\n    elif row.type == 'left_hand':\n        return row.landmark_index + 80\n    else:\n        return row.landmark_index\n\ndef visualise2d_landmarks(parquet_df, title=\"\"):\n    connections = [\n        [0, 1, 2, 3, 4],\n        [0, 5, 6, 7, 8],\n        [0, 9, 10, 11, 12],\n        [0, 13, 14, 15, 16],\n        [0, 17, 18, 19, 20],\n        [38, 36, 35, 34, 30, 31, 32, 33, 37],\n        [40, 39],\n        [52, 46, 50, 48, 46, 44, 42, 41, 43, 45, 47, 49, 45, 51],\n        [42, 54, 56, 58, 60, 62, 58],\n        [41, 53, 55, 57, 59, 61, 57],\n        [54, 53],\n        [80, 81, 82, 83, 84],\n        [80, 85, 86, 87, 88],\n        [80, 89, 90, 91, 92],\n        [80, 93, 94, 95, 96],\n        [80, 97, 98, 99, 100],\n    ]\n\n    parquet_df = map_new_to_old_style(parquet_df)\n    frames = sorted(set(parquet_df.frame))\n    first_frame = min(frames)\n\n    # Adding color and plot_order columns to the DataFrame\n    parquet_df['color'] = parquet_df.type.apply(lambda row: assign_color(row))\n    parquet_df['plot_order'] = parquet_df.apply(lambda row: assign_order(row), axis=1)\n    first_frame_df = parquet_df[parquet_df.frame == first_frame].copy()\n    first_frame_df = first_frame_df.sort_values([\"plot_order\"]).set_index('plot_order')\n\n    frames_l = []\n    for frame in frames:\n        filtered_df = parquet_df[parquet_df.frame == frame].copy()\n        filtered_df = filtered_df.sort_values([\"plot_order\"]).set_index(\"plot_order\")\n        traces = [go.Scatter(\n            x=filtered_df['x'],\n            y=filtered_df['y'],\n            mode='markers',\n            marker=dict(\n                color=filtered_df.color,\n                size=9\n            )\n        )]\n\n        for i, seg in enumerate(connections):\n            trace = go.Scatter(\n                x=filtered_df.loc[seg]['x'],\n                y=filtered_df.loc[seg]['y'],\n                mode='lines'\n            )\n            traces.append(trace)\n        frame_data = go.Frame(data=traces, traces=[i for i in range(17)])\n        frames_l.append(frame_data)\n\n    traces = [go.Scatter(\n        x=first_frame_df['x'],\n        y=first_frame_df['y'],\n        mode='markers',\n        marker=dict(\n            color=first_frame_df.color,\n            size=9\n        )\n    )]\n    for i, seg in enumerate(connections):\n        trace = go.Scatter(\n            x=first_frame_df.loc[seg]['x'],\n            y=first_frame_df.loc[seg]['y'],\n            mode='lines',\n            line=dict(\n                color='black',\n                width=2\n            )\n        )\n        traces.append(trace)\n    fig = go.Figure(\n        data=traces,\n        frames=frames_l\n    )\n\n    fig.update_layout(\n        width=500,\n        height=800,\n        scene={'aspectmode': 'data'},\n        updatemenus=[\n            {\n                \"buttons\": [\n                    {\n                        \"args\": [None, {\n                            \"frame\": {\"duration\": 100, \"redraw\": True},\n                            \"fromcurrent\": True,\n                            \"transition\": {\"duration\": 0}\n                        }],\n                        \"label\": \"&#9654;\",\n                        \"method\": \"animate\",\n                    },\n\n                ],\n                \"direction\": \"left\",\n                \"pad\": {\"r\": 100, \"t\": 100},\n                \"font\": {\"size\": 30},\n                \"type\": \"buttons\",\n                \"x\": 0.1,\n                \"y\": 0,\n            }\n        ],\n    )\n    camera = dict(\n        up=dict(x=0, y=-1, z=0),\n        eye=dict(x=0, y=0, z=2.5)\n    )\n    fig.update_layout(title_text=title, title_x=0.5)\n    fig.update_layout(scene_camera=camera, showlegend=False)\n    fig.update_layout(xaxis=dict(visible=False), yaxis=dict(visible=False))\n    fig.update_yaxes(autorange=\"reversed\")\n\n    fig.show()\n\ndef get_phrase(df, file_id, sequence_id):\n    '''\n    To get a phrase by filtering file_id and sequence_id at the train.csv file\n    Eg. file_id = 5414471, sequence_id = 1816796431 the phrase will be: '3 creekhouse'\n    '''\n    return df.loc[(df.file_id == file_id) & (df.sequence_id == sequence_id), 'phrase'].iloc[0]\n","metadata":{"execution":{"iopub.status.busy":"2023-06-30T19:03:33.000162Z","iopub.execute_input":"2023-06-30T19:03:33.000608Z","iopub.status.idle":"2023-06-30T19:03:33.037597Z","shell.execute_reply.started":"2023-06-30T19:03:33.000576Z","shell.execute_reply":"2023-06-30T19:03:33.036479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code reads a CSV file located at \"/kaggle/input/asl-fingerspelling/train.csv\" and stores the data in a DataFrame called df_train. The pd.read_csv() function from the pandas library is used to read the CSV file and return a DataFrame. Finally, the head() method is called on df_train to display the first few rows of the DataFrame, providing a glimpse of the data contained in the CSV file.","metadata":{}},{"cell_type":"code","source":"file_path = \"/kaggle/input/asl-fingerspelling/train.csv\"\ndf_train = pd.read_csv(file_path)\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-30T18:50:45.448331Z","iopub.execute_input":"2023-06-30T18:50:45.448757Z","iopub.status.idle":"2023-06-30T18:50:45.557542Z","shell.execute_reply.started":"2023-06-30T18:50:45.448729Z","shell.execute_reply":"2023-06-30T18:50:45.556343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe()","metadata":{"execution":{"iopub.status.busy":"2023-06-30T18:50:48.353799Z","iopub.execute_input":"2023-06-30T18:50:48.354957Z","iopub.status.idle":"2023-06-30T18:50:48.38218Z","shell.execute_reply.started":"2023-06-30T18:50:48.35492Z","shell.execute_reply":"2023-06-30T18:50:48.381165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The code snippet loads a Parquet file into a Pandas DataFrame. The file to be loaded is specified by the `file_id` variable, which is set to `5414471`, and the `sequence_id` variable, which is set to `1817123330`. \n\nThe Parquet file path is constructed using the `file_id` and stored in the `file_path_5414471` variable as \"/kaggle/input/asl-fingerspelling/train_landmarks/5414471.parquet\".\n\nThen, the Parquet file is read using the `pd.read_parquet()` function, passing the `file_path_5414471` as the argument. The resulting DataFrame is stored in the `df` variable.\n\nThis code is useful when you want to load a specific Parquet file into a DataFrame for further analysis or processing.\n","metadata":{}},{"cell_type":"code","source":"file_id = 5414471\nsequence_id = 1817123330\nfile_path_5414471 = \"/kaggle/input/asl-fingerspelling/train_landmarks/5414471.parquet\"\ndf = pd.read_parquet(file_path_5414471)","metadata":{"execution":{"iopub.status.busy":"2023-06-30T18:50:53.875008Z","iopub.execute_input":"2023-06-30T18:50:53.875448Z","iopub.status.idle":"2023-06-30T18:50:56.484538Z","shell.execute_reply.started":"2023-06-30T18:50:53.875413Z","shell.execute_reply":"2023-06-30T18:50:56.483547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-06-30T18:51:05.286533Z","iopub.execute_input":"2023-06-30T18:51:05.286935Z","iopub.status.idle":"2023-06-30T18:51:05.335153Z","shell.execute_reply.started":"2023-06-30T18:51:05.286904Z","shell.execute_reply":"2023-06-30T18:51:05.334025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(set(df.index))","metadata":{"execution":{"iopub.status.busy":"2023-06-30T18:51:08.914833Z","iopub.execute_input":"2023-06-30T18:51:08.91529Z","iopub.status.idle":"2023-06-30T18:51:08.951834Z","shell.execute_reply.started":"2023-06-30T18:51:08.915256Z","shell.execute_reply":"2023-06-30T18:51:08.950652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sequence = df[df.index == sequence_id]\ndf_sequence","metadata":{"execution":{"iopub.status.busy":"2023-06-30T18:51:11.4043Z","iopub.execute_input":"2023-06-30T18:51:11.406444Z","iopub.status.idle":"2023-06-30T18:51:11.447898Z","shell.execute_reply.started":"2023-06-30T18:51:11.406395Z","shell.execute_reply":"2023-06-30T18:51:11.446772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequence_phrase = get_phrase(df_train, \n                             file_id, \n                             sequence_id)\nsequence_phrase","metadata":{"execution":{"iopub.status.busy":"2023-06-30T18:51:19.919002Z","iopub.execute_input":"2023-06-30T18:51:19.919418Z","iopub.status.idle":"2023-06-30T18:51:19.928963Z","shell.execute_reply.started":"2023-06-30T18:51:19.919384Z","shell.execute_reply":"2023-06-30T18:51:19.927623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualise2d_landmarks(df_sequence, \n                      f\"Phrase: {sequence_phrase}\")","metadata":{"execution":{"iopub.status.busy":"2023-06-30T19:03:39.369469Z","iopub.execute_input":"2023-06-30T19:03:39.369845Z","iopub.status.idle":"2023-06-30T19:04:12.589054Z","shell.execute_reply.started":"2023-06-30T19:03:39.369818Z","shell.execute_reply":"2023-06-30T19:04:12.586809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[np.logical_and(df_train.file_id == file_id, df_train.sequence_id == sequence_id)].phrase.iloc[0]","metadata":{"execution":{"iopub.status.busy":"2023-06-30T19:07:02.602434Z","iopub.execute_input":"2023-06-30T19:07:02.602892Z","iopub.status.idle":"2023-06-30T19:07:02.613276Z","shell.execute_reply.started":"2023-06-30T19:07:02.602859Z","shell.execute_reply":"2023-06-30T19:07:02.612189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The code snippet `visualise2d_landmarks(df_sequence, f\"Phrase: {sequence_phrase}\")` is used to visualize 2D landmarks from a sequence dataframe `df_sequence`. It takes two parameters:\n\n- `df_sequence`: The sequence dataframe containing the data for landmarks.\n- `f\"Phrase: {sequence_phrase}\"`: The title of the visualization plot, which is generated by formatting the `sequence_phrase` variable into the string.\n\nThe function `visualise2d_landmarks` performs the following steps:\n1. Maps the sequence dataframe to the old style format using the `map_new_to_old_style` function.\n2. Extracts unique frames from the dataframe.\n3. Sets up the connections between landmarks for plotting.\n4. Assigns colors and plot order to each landmark based on their types.\n5. Creates a scatter plot for each frame, with markers representing the landmarks' coordinates.\n6. Connects the landmarks with lines according to the predefined connections.\n7. Animates the frames using the plotly library.\n8. Sets the layout and appearance of the plot, including width, height, aspect ratio, and buttons for animation control.\n9. Renders and displays the plot.\n\nThe resulting visualization provides a 2D representation of landmarks from the sequence dataframe, allowing for the analysis and exploration of the data. The title of the plot includes the corresponding phrase for reference.\n","metadata":{}},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-06-30T19:18:24.631675Z","iopub.execute_input":"2023-06-30T19:18:24.63211Z","iopub.status.idle":"2023-06-30T19:18:24.638799Z","shell.execute_reply.started":"2023-06-30T19:18:24.632068Z","shell.execute_reply":"2023-06-30T19:18:24.637411Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## III. Suplemental data  \nSimilar steps to the train data as above","metadata":{}},{"cell_type":"code","source":"file_path = \"/kaggle/input/asl-fingerspelling/supplemental_metadata.csv\"\ndf_supp = pd.read_csv(file_path)\ndf_supp.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-30T19:20:00.256643Z","iopub.execute_input":"2023-06-30T19:20:00.257031Z","iopub.status.idle":"2023-06-30T19:20:00.366864Z","shell.execute_reply.started":"2023-06-30T19:20:00.257003Z","shell.execute_reply":"2023-06-30T19:20:00.365718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_supp.describe()","metadata":{"execution":{"iopub.status.busy":"2023-06-30T19:20:18.88719Z","iopub.execute_input":"2023-06-30T19:20:18.887585Z","iopub.status.idle":"2023-06-30T19:20:18.914547Z","shell.execute_reply.started":"2023-06-30T19:20:18.887556Z","shell.execute_reply":"2023-06-30T19:20:18.913427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_id = 33432165\nsequence_id = 1535467051\nfile_path_33432165 = \"/kaggle/input/asl-fingerspelling/supplemental_landmarks/33432165.parquet\"\ndf = pd.read_parquet(file_path_33432165)","metadata":{"execution":{"iopub.status.busy":"2023-06-30T19:35:13.268779Z","iopub.execute_input":"2023-06-30T19:35:13.269256Z","iopub.status.idle":"2023-06-30T19:35:27.562276Z","shell.execute_reply.started":"2023-06-30T19:35:13.269222Z","shell.execute_reply":"2023-06-30T19:35:27.561331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-06-30T19:37:31.735141Z","iopub.execute_input":"2023-06-30T19:37:31.737818Z","iopub.status.idle":"2023-06-30T19:37:31.813759Z","shell.execute_reply.started":"2023-06-30T19:37:31.737741Z","shell.execute_reply":"2023-06-30T19:37:31.812459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### number of unique sequences = 1000","metadata":{}},{"cell_type":"code","source":"len(set(df.index))","metadata":{"execution":{"iopub.status.busy":"2023-06-30T19:37:37.304779Z","iopub.execute_input":"2023-06-30T19:37:37.305974Z","iopub.status.idle":"2023-06-30T19:37:37.343529Z","shell.execute_reply.started":"2023-06-30T19:37:37.305926Z","shell.execute_reply":"2023-06-30T19:37:37.34207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sequence = df[df.index == sequence_id]\ndf_sequence","metadata":{"execution":{"iopub.status.busy":"2023-06-30T19:37:47.448297Z","iopub.execute_input":"2023-06-30T19:37:47.449351Z","iopub.status.idle":"2023-06-30T19:37:47.488389Z","shell.execute_reply.started":"2023-06-30T19:37:47.449315Z","shell.execute_reply":"2023-06-30T19:37:47.487164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequence_phrase = get_phrase(df_supp, \n                             file_id, \n                             sequence_id)\nsequence_phrase","metadata":{"execution":{"iopub.status.busy":"2023-06-30T19:37:51.403901Z","iopub.execute_input":"2023-06-30T19:37:51.404305Z","iopub.status.idle":"2023-06-30T19:37:51.413508Z","shell.execute_reply.started":"2023-06-30T19:37:51.404275Z","shell.execute_reply":"2023-06-30T19:37:51.412414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualise2d_landmarks(df_sequence, \n                      f\"Phrase: {sequence_phrase}\")","metadata":{"execution":{"iopub.status.busy":"2023-06-30T19:38:33.208505Z","iopub.execute_input":"2023-06-30T19:38:33.208929Z","iopub.status.idle":"2023-06-30T19:38:53.580296Z","shell.execute_reply.started":"2023-06-30T19:38:33.208896Z","shell.execute_reply":"2023-06-30T19:38:53.578471Z"},"trusted":true},"execution_count":null,"outputs":[]}]}