{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"(imports and functions)","metadata":{}},{"cell_type":"code","source":"\n### import libraries\nimport pandas as pd,numpy as np,os\nimport json\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.io as pio\nfrom pathlib import Path\n\nfrom sklearn.decomposition import PCA\nfrom sklearn.cluster import KMeans\nfrom matplotlib.colors import ListedColormap\nfrom sklearn.preprocessing import normalize\n\nprint(\"importing\")","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-05-23T10:24:57.680471Z","iopub.execute_input":"2023-05-23T10:24:57.681164Z","iopub.status.idle":"2023-05-23T10:25:00.816148Z","shell.execute_reply.started":"2023-05-23T10:24:57.681113Z","shell.execute_reply":"2023-05-23T10:25:00.814708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def map_new_to_old_style(sequence):\n    types = []\n    landmark_indexes = []\n    for column in list(sequence.columns)[1:544]:\n        parts = column.split(\"_\")\n        if len(parts) == 4:\n            types.append(parts[1] + \"_\" + parts[2])\n        else:\n            types.append(parts[1])\n\n        landmark_indexes.append(int(parts[-1]))\n\n    data = {\n        \"frame\": [],\n        \"type\": [],\n        \"landmark_index\": [],\n        \"x\": [],\n        \"y\": [],\n        \"z\": []\n    }\n\n    for index, row in sequence.iterrows():\n        data[\"frame\"] += [int(row.frame)]*543\n        data[\"type\"] += types\n        data[\"landmark_index\"] += landmark_indexes\n\n        for _type, landmark_index in zip(types, landmark_indexes):\n            data[\"x\"].append(row[f\"x_{_type}_{landmark_index}\"])\n            data[\"y\"].append(row[f\"y_{_type}_{landmark_index}\"])\n            data[\"z\"].append(row[f\"z_{_type}_{landmark_index}\"])\n\n    return pd.DataFrame.from_dict(data)\n\n# assign desired colors to landmarks\ndef assign_color(row):\n    if row == 'face':\n        return 'red'\n    elif 'hand' in row:\n        return 'dodgerblue'\n    else:\n        return 'green'\n\n# specifies the plotting order\ndef assign_order(row):\n    if row.type == 'face':\n        return row.landmark_index + 101\n    elif row.type == 'pose':\n        return row.landmark_index + 30\n    elif row.type == 'left_hand':\n        return row.landmark_index + 80\n    else:\n        return row.landmark_index\n    \ndef visualise2d_landmarks(parquet_df, title=\"\",inter_frame_delay : int = 100):\n    connections = [  \n        [0, 1, 2, 3, 4,],\n        [0, 5, 6, 7, 8],\n        [0, 9, 10, 11, 12],\n        [0, 13, 14, 15, 16],\n        [0, 17, 18, 19, 20],\n\n        \n        [38, 36, 35, 34, 30, 31, 32, 33, 37],\n        [40, 39],\n        [52, 46, 50, 48, 46, 44, 42, 41, 43, 45, 47, 49, 45, 51],\n        [42, 54, 56, 58, 60, 62, 58],\n        [41, 53, 55, 57, 59, 61, 57],\n        [54, 53],\n\n        \n        [80, 81, 82, 83, 84, ],\n        [80, 85, 86, 87, 88],\n        [80, 89, 90, 91, 92],\n        [80, 93, 94, 95, 96],\n        [80, 97, 98, 99, 100], ]\n\n    parquet_df = map_new_to_old_style(parquet_df)\n    frames = sorted(set(parquet_df.frame))\n    first_frame = min(frames)\n    parquet_df['color'] = parquet_df.type.apply(lambda row: assign_color(row))\n    parquet_df['plot_order'] = parquet_df.apply(lambda row: assign_order(row), axis=1)\n    first_frame_df = parquet_df[parquet_df.frame == first_frame].copy()\n    first_frame_df = first_frame_df.sort_values([\"plot_order\"]).set_index('plot_order')\n\n\n    frames_l = []\n    for frame in frames:\n        filtered_df = parquet_df[parquet_df.frame == frame].copy()\n        filtered_df = filtered_df.sort_values([\"plot_order\"]).set_index(\"plot_order\")\n        traces = [go.Scatter(\n            x=filtered_df['x'],\n            y=filtered_df['y'],\n            mode='markers',\n            marker=dict(\n                color=filtered_df.color,\n                size=9))]\n\n        for i, seg in enumerate(connections):\n            trace = go.Scatter(\n                    x=filtered_df.loc[seg]['x'],\n                    y=filtered_df.loc[seg]['y'],\n                    mode='lines',\n            )\n            traces.append(trace)\n        frame_data = go.Frame(data=traces, traces = [i for i in range(17)])\n        frames_l.append(frame_data)\n\n    traces = [go.Scatter(\n        x=first_frame_df['x'],\n        y=first_frame_df['y'],\n        mode='markers',\n        marker=dict(\n            color=first_frame_df.color,\n            size=9\n        )\n    )]\n    for i, seg in enumerate(connections):\n        trace = go.Scatter(\n            x=first_frame_df.loc[seg]['x'],\n            y=first_frame_df.loc[seg]['y'],\n            mode='lines',\n            line=dict(\n                color='black',\n                width=2\n            )\n        )\n        traces.append(trace)\n    \n    fig = go.Figure(\n            data=traces,\n            frames=frames_l\n        )\n\n    fig.update_layout(\n        width=500,\n        height=800,\n        scene={\n            'aspectmode': 'data',\n        },\n        updatemenus=[\n            {\n                \"buttons\": [\n                    {\n                        \"args\": [None, {\"frame\": {\"duration\": inter_frame_delay,\n                                                  \"redraw\": True},\n                                        \"fromcurrent\": True,\n                                        \"transition\": {\"duration\": 0}}],\n                        \"label\": \"&#9654;\",\n                        \"method\": \"animate\",\n                    },\n                    {\n                        \"args\": [[None], {\"frame\": {\"duration\": 0, \"redraw\": False},\n                                          \"mode\": \"immediate\",\n                                          \"transition\": {\"duration\": 0}}],\n                        \"label\": \"&#9612;&#9612;\",\n                        \"method\": \"animate\",\n                    },\n                ],\n                \"direction\": \"left\",\n                \"pad\": {\"r\": 100, \"t\": 100},\n                \"font\": {\"size\":20},\n                \"type\": \"buttons\",\n                \"x\": 0.1,\n                \"y\": 0,\n            }\n        ],\n    )\n    camera = dict(\n        up=dict(x=0, y=-1, z=0),\n        eye=dict(x=0, y=0, z=2.5)\n    )\n    fig.update_layout(title_text=title, title_x=0.5)\n    fig.update_layout(scene_camera=camera, showlegend=False)\n    fig.update_layout(xaxis = dict(visible=False),\n            yaxis = dict(visible=False),\n    )\n    fig.update_yaxes(autorange=\"reversed\")\n\n    fig.show()\n    \n    \ndef get_phrase(df, file_id, sequence_id):\n    return df[\n        np.logical_and(\n            df.file_id == file_id, \n            df.sequence_id == sequence_id\n        )\n    ].phrase.iloc[0]\n","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-05-23T10:25:00.821464Z","iopub.execute_input":"2023-05-23T10:25:00.821911Z","iopub.status.idle":"2023-05-23T10:25:00.859007Z","shell.execute_reply.started":"2023-05-23T10:25:00.821874Z","shell.execute_reply":"2023-05-23T10:25:00.857503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Goal of this notebook\n\nThe aim of this notebook is to help me to:\n* Understand the representation of the signs in the dataset,\n* Get something simple that I can train a basic neural network with.\n\nThe ideal outcome would be series of examples of \"this represents 'a'\", \"this is  a 'b'\"","metadata":{}},{"cell_type":"markdown","source":"## GOAL OF COMPETITION\n\nThe goal of this competition is to detect and translate American Sign Language (ASL) fingerspelling into text. You will create a model trained on the largest dataset of its kind, released specifically for this competition. The data includes more than three million fingerspelled characters produced by over 100 Deaf signers captured via the selfie camera of a smartphone with a variety of backgrounds and lighting conditions\n\n* Your work may help move sign language recognition forward, making AI more accessible for the Deaf and Hard of Hearing community. ","metadata":{}},{"cell_type":"markdown","source":"<img src=\"https://www.lifeprint.com/asl101/fingerspelling/images/signlanguageabc.jpg\" width=\"400\">","metadata":{}},{"cell_type":"markdown","source":"# Exploring the training data\n\nFirst let's read in and inspect the training data legend or key:","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv(\"/kaggle/input/asl-fingerspelling/train.csv\", delimiter=',', encoding='UTF-8')\npd.set_option('display.max_columns', None)\ndata.head(3)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:00.865679Z","iopub.execute_input":"2023-05-23T10:25:00.866157Z","iopub.status.idle":"2023-05-23T10:25:01.105914Z","shell.execute_reply.started":"2023-05-23T10:25:00.866108Z","shell.execute_reply":"2023-05-23T10:25:01.104696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's try to visualise the first phrase in the training data `3 creekhouse` - which is sequence ID `1816796431`\n\nApparently it's somewhere in `train_landmarks/5414471.parquet`","metadata":{}},{"cell_type":"code","source":"phrase_string = data.loc[data[\"sequence_id\"] == 1816796431].phrase.values[0]\n\nphrases = pd.read_parquet(\"/kaggle/input/asl-fingerspelling/train_landmarks/5414471.parquet\")\nphrases.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:01.107617Z","iopub.execute_input":"2023-05-23T10:25:01.108376Z","iopub.status.idle":"2023-05-23T10:25:18.421504Z","shell.execute_reply.started":"2023-05-23T10:25:01.108328Z","shell.execute_reply":"2023-05-23T10:25:18.420636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The parquet is a sequence of co-oridinates, organised by the video frame and the phrase the person was signing.\n\nLet's get out a phrase to inspect in more detail, so we can mess with that in isolation:","metadata":{}},{"cell_type":"code","source":"target_phrase = phrases.loc[1816796431]\ntarget_phrase.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:18.422713Z","iopub.execute_input":"2023-05-23T10:25:18.423577Z","iopub.status.idle":"2023-05-23T10:25:19.992083Z","shell.execute_reply.started":"2023-05-23T10:25:18.42354Z","shell.execute_reply":"2023-05-23T10:25:19.990833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The phrase itself then is the co-ordinates of the person signing our phase, and can be animated with a [complex function I've shamelessly stolen from another user](https://www.kaggle.com/code/sharifi76/eda-motion-visualization-fingerspelling)","metadata":{}},{"cell_type":"code","source":"visualise2d_landmarks(target_phrase, f\"Phrase: {phrase_string}\", inter_frame_delay=200)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:19.993897Z","iopub.execute_input":"2023-05-23T10:25:19.994263Z","iopub.status.idle":"2023-05-23T10:25:40.100484Z","shell.execute_reply.started":"2023-05-23T10:25:19.994229Z","shell.execute_reply":"2023-05-23T10:25:40.098793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"OK, so we have some frames that correspond to the phrase. Let's look at the hands specifically:","metadata":{}},{"cell_type":"code","source":"target_phrase.head(10).filter(regex=\"hand|frame\")","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:40.102265Z","iopub.execute_input":"2023-05-23T10:25:40.103061Z","iopub.status.idle":"2023-05-23T10:25:40.239961Z","shell.execute_reply.started":"2023-05-23T10:25:40.103008Z","shell.execute_reply":"2023-05-23T10:25:40.23875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Mostly we have no hand co-ordinates, which matches our prior visualisation.\n\nCan we detect pauses or similar so we can guess at letters?","metadata":{}},{"cell_type":"code","source":"#let's first get rid of the useless columns - as our speller should be using only one hand\nhands = target_phrase.filter(regex=\"hand|frame\").copy()\nhands = hands.dropna(axis=1,how=\"all\")\n\n#then add a \"group\" column to see how many \"pauses\" we have where there's no detection\n\nhands[\"group_no\"] = hands.filter(regex=\"hand\").isna().all(axis=1).cumsum()\n\n#we can then drop the \"pauses\", using thresholds as we've no longer got a whole row of nans\nhands = hands.dropna(axis=0,thresh=20)\nhands.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:40.241883Z","iopub.execute_input":"2023-05-23T10:25:40.242371Z","iopub.status.idle":"2023-05-23T10:25:40.319949Z","shell.execute_reply.started":"2023-05-23T10:25:40.242327Z","shell.execute_reply":"2023-05-23T10:25:40.318671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"letter_groups = hands.group_no.nunique()\nprint(f\"detected {letter_groups} separate signs.\")\nprint(f\"The target phrase '3 creekhouse' has {len('3 creekhouse')} letters\")\nprint(f\"The target phrase '3 creekhouse' has {len(set('3 creekhouse'))} unique letters\")","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:40.32457Z","iopub.execute_input":"2023-05-23T10:25:40.324956Z","iopub.status.idle":"2023-05-23T10:25:40.334886Z","shell.execute_reply.started":"2023-05-23T10:25:40.324922Z","shell.execute_reply":"2023-05-23T10:25:40.333672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Obviously this method of detecting which sign matches which letter will not work. \n\n# Clustering to extract single letter signs\n\nwe know there should be 9 unique letters (igoring the space), if we treat each dataframe row as an embedding, can we cluster the data into 9 broad groups?\n\nFirst let's do a scree plot to find the dimensionality for PCA for the clustering","metadata":{}},{"cell_type":"code","source":"feat = hands.filter(regex=\"hand\").to_numpy()\nfeat = normalize(feat)\nn_components = min(feat.shape)\npca  = PCA(n_components=n_components)\npca.fit(feat)\n\nexplained_variance = pca.explained_variance_\ncumulative_explained_variance = np.cumsum(explained_variance)\nplt.figure(figsize=(8, 6))\nplt.plot(range(1, n_components + 1), explained_variance, marker='o', linestyle='-', label='Explained Variance')\nplt.plot(range(1, len(cumulative_explained_variance) + 1), cumulative_explained_variance, marker='o', linestyle='-', label='Cumulative Explained Variance')\n\nplt.xlabel('Principal Component')\nplt.ylabel('Explained Variance')\nplt.title('Scree Plot')\nplt.legend()\nplt.grid()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:40.336402Z","iopub.execute_input":"2023-05-23T10:25:40.336803Z","iopub.status.idle":"2023-05-23T10:25:40.793125Z","shell.execute_reply.started":"2023-05-23T10:25:40.336769Z","shell.execute_reply":"2023-05-23T10:25:40.791916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks like this is 10 dimensional","metadata":{}},{"cell_type":"code","source":"pca = PCA(n_components=10)\npca.fit(feat)\nx = pca.transform(feat)\n\nkmeans = KMeans(n_clusters=9,n_init=\"auto\")\nkmeans.fit(x)\n\nunique_labels = np.unique(kmeans.labels_)\ncmap = ListedColormap(plt.cm.rainbow(np.linspace(0, 1, len(unique_labels))))\n\nplt.scatter(x[:, 0], x[:, 1], c=kmeans.labels_, cmap=cmap)\nplt.title(\"K-Means Clustering (first two eigenvectors of 10)\")\n\n\nhandles = [plt.Line2D([], [], linestyle='', marker='o', markersize=2, color=cmap(i), label=f'Cluster {label}') for i, label in enumerate(unique_labels)]\nplt.legend(handles=handles)\n\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:40.794654Z","iopub.execute_input":"2023-05-23T10:25:40.795111Z","iopub.status.idle":"2023-05-23T10:25:41.255281Z","shell.execute_reply.started":"2023-05-23T10:25:40.795081Z","shell.execute_reply":"2023-05-23T10:25:41.254299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks kind of clustered. Maybe.\n\nLet's see what it says these signs look like:","metadata":{}},{"cell_type":"code","source":"hands[\"Sign Group\"] = kmeans.labels_\n    \nhands.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:41.256626Z","iopub.execute_input":"2023-05-23T10:25:41.257379Z","iopub.status.idle":"2023-05-23T10:25:41.322017Z","shell.execute_reply.started":"2023-05-23T10:25:41.257345Z","shell.execute_reply":"2023-05-23T10:25:41.320774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"OK, so we have the cluster label for each \"letter\" in the sequence. Let's see if that makes broad sense given:\n* Our sequence should say \"3 creekhouse\"\n* We should have different groupings for the first 4 symbols\n* The 4th symbol should be the same as the last.","metadata":{}},{"cell_type":"code","source":"print(\"all labels\")\nprint(kmeans.labels_)\ncipher = []\n#let's only keep labels found in consecutive detections - discarding others as noise\nfor i, element in enumerate(kmeans.labels_):\n    if len(cipher) == 0 or (i+1 < len(kmeans.labels_) and element == kmeans.labels_[i+1] and element != cipher[-1]):\n        cipher.append(element)\n        \n\nprint(\"Repeated labels\")\nprint(cipher)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:41.323842Z","iopub.execute_input":"2023-05-23T10:25:41.324196Z","iopub.status.idle":"2023-05-23T10:25:41.33193Z","shell.execute_reply.started":"2023-05-23T10:25:41.324166Z","shell.execute_reply":"2023-05-23T10:25:41.330669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So now we have our crude cipher,but there's no sybol at the beginning that's not at the end... it's a bit of a jumbled up mess. If we  try the simple substitution translation we get:","metadata":{}},{"cell_type":"code","source":"unique_string = \"\"\nfor value in cipher:\n    value = str(value)\n    if value not in unique_string:\n        unique_string += value\n    else:\n        break\nprint(unique_string)\nunrepeated_phrase = \"\".join(dict.fromkeys(phrase_string.replace(\" \",\"\")))\nplain_subtext =  unrepeated_phrase[:(len(unique_string))]\nprint(plain_subtext)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:41.333332Z","iopub.execute_input":"2023-05-23T10:25:41.333735Z","iopub.status.idle":"2023-05-23T10:25:41.347011Z","shell.execute_reply.started":"2023-05-23T10:25:41.3337Z","shell.execute_reply":"2023-05-23T10:25:41.345879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#can't use the whole of the cipher string here - as the repetitions kick in wrongly after the '5'\ntrans_table = str.maketrans(unique_string, plain_subtext)\ncipherstring = [str(element) for element in cipher]   \ndecoded_word = ''.join(cipherstring).translate(trans_table)\nprint(f\"We're supposd to get: {unrepeated_phrase}\")\nprint(f\"We got:               {decoded_word}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:41.348506Z","iopub.execute_input":"2023-05-23T10:25:41.348899Z","iopub.status.idle":"2023-05-23T10:25:41.359504Z","shell.execute_reply.started":"2023-05-23T10:25:41.348867Z","shell.execute_reply":"2023-05-23T10:25:41.358406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Not terrible...\n\nThe `.translate` method though appears to sometimes double translate the cipher - if the first group is a `3` that translates to a `2`, but we later make `2` tranlsate to `g` we get a `g` in both places. This in mind I've written a more strict translation algorithm that will perform more slowly, but whose behaviour I can fully control:","metadata":{}},{"cell_type":"code","source":"def strict_translate(encoded_list,  known_letters , debug:bool = False):\n    my_encoded_list = encoded_list.copy()\n    encoded_fragment = [str(element) for element in my_encoded_list[:len(known_letters)].copy()]\n    \n    for find, replace in zip(encoded_fragment,known_letters):\n        for i,element in enumerate(my_encoded_list):\n            if not isinstance(element, str) and str(element) == find:\n                if debug:\n                    print(f\"replacing {element} with {replace}\")\n                    \n                my_encoded_list[i] = replace\n    \n    decoded_text = ''.join([str(element) for element in my_encoded_list])\n    return decoded_text","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:41.3608Z","iopub.execute_input":"2023-05-23T10:25:41.361134Z","iopub.status.idle":"2023-05-23T10:25:41.373089Z","shell.execute_reply.started":"2023-05-23T10:25:41.361105Z","shell.execute_reply":"2023-05-23T10:25:41.37187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(type(cipher))\nprint(strict_translate(cipher,plain_subtext))\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:41.37499Z","iopub.execute_input":"2023-05-23T10:25:41.375342Z","iopub.status.idle":"2023-05-23T10:25:41.390051Z","shell.execute_reply.started":"2023-05-23T10:25:41.375311Z","shell.execute_reply":"2023-05-23T10:25:41.388767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Is this repeatable?\n\nFirst thing to do is to put the the clustering and \"translation\" into a function so we can pass it other samples","metadata":{}},{"cell_type":"code","source":"def try_clustering(sample_id: int, data : pd.DataFrame, savethreshold = 80):\n        #get the text that's being signed\n        phrase_string = data.loc[data[\"sequence_id\"] == sample_id].phrase.values[0]\n        \n        #where's the \"movie\" stored?\n        filename = data.loc[data[\"sequence_id\"] == sample_id].file_id.values[0]\n        \n        #get the movie itself\n        target_phrase = pd.read_parquet(f\"/kaggle/input/asl-fingerspelling/train_landmarks/{filename}.parquet\").loc[sample_id]\n        \n        #delete everything but the detected hand, and the frame number\n        hand = target_phrase.filter(regex=\"hand|frame\").copy().dropna(axis=1,how=\"all\")\n        \n        #remove frames where there's very few hand points detected\n        hand = hand.dropna(axis=0,thresh=20)\n        \n        #find the dimensionality of the dataset:\n        feat = hand.filter(regex=\"hand\").to_numpy()\n        feat = normalize(np.nan_to_num(feat))\n        n_components = min(feat.shape)\n        pca  = PCA(n_components=n_components)\n        pca.fit(feat)\n\n        explained_variance = pca.explained_variance_\n        \n        #work out a cut-off threshold for PCA  - the last dimension that matters.\n        #in this case I've selected the dimension that contributes 1% as much information as the first one - or 50 if that fails.\n        n_components = next((i for i, dimension in enumerate(explained_variance) if dimension / explained_variance[0] <= 0.01), 50)\n\n        pca = PCA(n_components=n_components)\n        pca.fit(feat)\n        x = pca.transform(feat)\n        \n        # find unique letters (less one for \" \" space)\n        unique_letters = len(set(phrase_string)) -1 \n        \n        kmeans = KMeans(n_clusters=unique_letters,n_init=\"auto\")\n        \n        try:\n        \n            kmeans.fit(x)\n        except:\n            #most likely there's not enough \"non-blank\" frames to cluster them into letters - that means the video should probably be discarded\n            return 0\n        \n        cipher = []\n        #let's only keep labels found in consecutive detections - discarding others as noise\n        for i, element in enumerate(kmeans.labels_):\n            if len(cipher) == 0 or (i+1 < len(kmeans.labels_) and element == kmeans.labels_[i+1] and element != cipher[-1]):\n                cipher.append(element)\n\n        cipherstr = [str(element) for element in cipher]\n        \n        unique_string = \"\"\n        for value in cipherstr:\n            if value not in unique_string:\n                unique_string += value\n            else:\n                break\n        \n        unrepeated_phrase = no_repeats(phrase_string.replace(\" \",\"\"))\n        \n        plain_subtext =  unrepeated_phrase[:(len(unique_string))]\n        \n        \n        decoded_word = strict_translate(cipher,plain_subtext)\n        \n        print(f\"We're supposed to get: {unrepeated_phrase}\")\n        print(f\"We got:                {decoded_word}\")\n        this_sim = similarity(unrepeated_phrase,decoded_word)\n        print(f\"This is a { this_sim : .2f}% similarity.\")\n        if this_sim >= savethreshold:\n            try:\n                save_gestures(hand,kmeans.labels_,cipher,unrepeated_phrase,decoded_word,sample_id)\n            except:\n                ... #having some issues saving if there's not enough detected signs\n        return this_sim\n        \ndef no_repeats(text: str): # removes consecutive repeated letters\n    return \"\".join(dict.fromkeys(text))\n\ndef similarity(string1 : str, string2 : str):\n    hits = 0\n    for i, let in enumerate(string1):\n        try:\n            if string2[i] == let:\n                hits += 1\n        except:\n            ... # the two strings are not the same length, so there's no \"hit\" in this context\n    \n    return 100 * hits / len(string1)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:41.391708Z","iopub.execute_input":"2023-05-23T10:25:41.392485Z","iopub.status.idle":"2023-05-23T10:25:41.414703Z","shell.execute_reply.started":"2023-05-23T10:25:41.392406Z","shell.execute_reply":"2023-05-23T10:25:41.413704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"That might work...","metadata":{}},{"cell_type":"code","source":"import pickle\ndef save_gestures(hand_df,groups,cipher, unrepeated_phrase,decoded_word,sequence_id):\n    #print(f\"writing {sequence_id} Pickle\")\n    gestures = {}\n    \n    for i, letter in enumerate(unrepeated_phrase):\n        if letter == decoded_word[i]:\n            cluster = cipher[i]\n            #print(f\"Letter: {letter}, Cluster: {cluster}\")\n            for i,group in enumerate(groups):\n                \n                if group == cluster:\n                    if letter in gestures:\n                        glist = gestures[letter]\n                        glist.append({\"Sequence\": sequence_id, \"Frame\": hand_df.iloc[i].frame})\n                        gestures[letter] = glist\n                    else:\n                        gList = [{\"Sequence\": sequence_id, \"Frame\": hand_df.iloc[i].frame}]\n                        gestures[letter] = gList\n    \n    with open(f'{sequence_id}.pickle', 'wb') as f:\n        pickle.dump(gestures, f)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:41.415784Z","iopub.execute_input":"2023-05-23T10:25:41.416154Z","iopub.status.idle":"2023-05-23T10:25:41.430938Z","shell.execute_reply.started":"2023-05-23T10:25:41.416121Z","shell.execute_reply":"2023-05-23T10:25:41.429504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive':\n    n_samples= 5\n    similarity_threshold = 40 # just so we get some output!\nelse:\n    n_samples = 20\n    similarity_threshold = 75 # we can be fussier here as we don't have to sit around waiting!\n    \ncumulative_similarity = 0\nfor sequence in data.sample(n_samples).sequence_id:\n    print(f\"Sequence ID: {sequence}\")\n    cumulative_similarity += try_clustering(sequence, data,similarity_threshold)\n    \nprint(f\"Average Similarity was: {cumulative_similarity/n_samples : .2f}%\")","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:25:41.432424Z","iopub.execute_input":"2023-05-23T10:25:41.433411Z","iopub.status.idle":"2023-05-23T10:26:59.405431Z","shell.execute_reply.started":"2023-05-23T10:25:41.433369Z","shell.execute_reply":"2023-05-23T10:26:59.399728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Useful format for signs data\nThe above file-writing method is great for a notebook, and is also saving our progress in a really visible way as we process more video sections, but frankly it's a bit rubbish when you want to start using it to make a model. With that in mind let's re-process the outputted files into a big old dataframe to make our future work easier: ","metadata":{}},{"cell_type":"code","source":"import pathlib\n\nfrom concurrent.futures import ThreadPoolExecutor\nimport itertools\n\ndef load_pickle_list(thisPickle):\n    with open(str(thisPickle), 'rb') as handle:\n        batch_signs = pickle.load(handle)\n\n    flattened_data = []\n    for letter, dicts in batch_signs.items():\n        for dict_ in dicts:\n            flattened_data.append({\n                'Letter': letter,\n                'Frame': dict_['Frame'],\n                'Sequence': dict_['Sequence'],\n            })\n    return flattened_data\n\npickleFolder = pathlib.Path(\".\")\nsign_pickles = pickleFolder.rglob('*.pickle')\nlist_of_pickles =  [str(p) for p in sign_pickles]\n\nwith ThreadPoolExecutor() as executor:\n    filesList = list(itertools.chain.from_iterable(list(executor.map(load_pickle_list, list_of_pickles))))\n\ndf = pd.DataFrame(filesList)\n\ndf.to_parquet(\"combined.parquet\", engine = 'pyarrow', compression = 'gzip')\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T10:26:59.407316Z","iopub.execute_input":"2023-05-23T10:26:59.40778Z","iopub.status.idle":"2023-05-23T10:26:59.465805Z","shell.execute_reply.started":"2023-05-23T10:26:59.407736Z","shell.execute_reply":"2023-05-23T10:26:59.464214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## _________________________THANK YOU !!! ___________________\n\nplease upvote if you like my work.","metadata":{}}]}