{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-24T09:00:40.641274Z","iopub.execute_input":"2023-06-24T09:00:40.641637Z","iopub.status.idle":"2023-06-24T09:00:40.668561Z","shell.execute_reply.started":"2023-06-24T09:00:40.641609Z","shell.execute_reply":"2023-06-24T09:00:40.667487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sign Language Recognition Challenge\n\nThe goal of this competition is to classify American Sign Language (ASL) signs.\n\nThe landmarks were extracted from raw videos with the MediaPipe holistic model and are asked to predict the sign from this data.","metadata":{}},{"cell_type":"markdown","source":"# 1. DATA OVERVIEW \n\nIn this notebook, we will be exploring and analyzing the dataset. We will start by importing important libraries, loading the data, and giving a brief description of the dataset.","metadata":{}},{"cell_type":"markdown","source":"**1.1 Import Libraries** ","metadata":{}},{"cell_type":"code","source":"import gc\n\nimport json\nfrom tqdm import tqdm\n\nimport torch\nimport torch.nn as nn\n\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\nimport warnings\nwarnings.filterwarnings(action='ignore')","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:00:46.479154Z","iopub.execute_input":"2023-06-24T09:00:46.479508Z","iopub.status.idle":"2023-06-24T09:00:46.48527Z","shell.execute_reply.started":"2023-06-24T09:00:46.47948Z","shell.execute_reply":"2023-06-24T09:00:46.484176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import data processing and visualisation libraries\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport plotly.io as pio\npio.templates.default = \"simple_white\"\n\n# import tensorflow and keras\nimport tensorflow as tf\nimport os\n\nprint(\"Packages imported...\")","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:00:52.434621Z","iopub.execute_input":"2023-06-24T09:00:52.435872Z","iopub.status.idle":"2023-06-24T09:01:00.876883Z","shell.execute_reply.started":"2023-06-24T09:00:52.435828Z","shell.execute_reply":"2023-06-24T09:01:00.875821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All the important packages have been loaded. Now, we'll load the training and supplemental_metadata dataframe.","metadata":{}},{"cell_type":"markdown","source":"**1.2 Loading Dataset** ","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/asl-fingerspelling/train.csv\")\nmetadata = pd.read_csv(\"/kaggle/input/asl-fingerspelling/supplemental_metadata.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:01:04.789915Z","iopub.execute_input":"2023-06-24T09:01:04.790786Z","iopub.status.idle":"2023-06-24T09:01:04.990493Z","shell.execute_reply.started":"2023-06-24T09:01:04.790749Z","shell.execute_reply":"2023-06-24T09:01:04.989638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**1.3 Data Description**","metadata":{}},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:01:08.296788Z","iopub.execute_input":"2023-06-24T09:01:08.297148Z","iopub.status.idle":"2023-06-24T09:01:08.321362Z","shell.execute_reply.started":"2023-06-24T09:01:08.297119Z","shell.execute_reply":"2023-06-24T09:01:08.320483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The phrases in the training set contains random websites/addresses/phone numbers","metadata":{}},{"cell_type":"code","source":"print(f\"Total number of files : {df.shape[0]}\")\nprint(f\"Total number of Participant in the dataset : {df.participant_id.nunique()}\")\nprint(f\"Total number of unique phrases : {df.phrase.nunique()}\")","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:01:12.060462Z","iopub.execute_input":"2023-06-24T09:01:12.060845Z","iopub.status.idle":"2023-06-24T09:01:12.084595Z","shell.execute_reply.started":"2023-06-24T09:01:12.060816Z","shell.execute_reply":"2023-06-24T09:01:12.083575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:01:15.102722Z","iopub.execute_input":"2023-06-24T09:01:15.103454Z","iopub.status.idle":"2023-06-24T09:01:15.113726Z","shell.execute_reply.started":"2023-06-24T09:01:15.103418Z","shell.execute_reply":"2023-06-24T09:01:15.11272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The metadata consists of normal sentences only..","metadata":{}},{"cell_type":"markdown","source":"**The data consist of:**\npath - The path to the landmark file.\n\nfile_id - A unique identifier for the data file.\n\nparticipant_id - A unique identifier for the data contributor.\n\nsequence_id - A unique identifier for the landmark sequence. Each data file may contain many sequences.\n\nphrase - The labels for the landmark sequence. The train and test datasets contain randomly generated addresses, phone numbers, and urls derived from components of real addresses/phone numbers/urls.","metadata":{}},{"cell_type":"markdown","source":"Let's checkout Dataframe information more...","metadata":{}},{"cell_type":"code","source":"# Check the dimensions of the dataset\ndf.shape\n\n# Check the data types of columns\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:01:19.855037Z","iopub.execute_input":"2023-06-24T09:01:19.856066Z","iopub.status.idle":"2023-06-24T09:01:19.895979Z","shell.execute_reply.started":"2023-06-24T09:01:19.856028Z","shell.execute_reply":"2023-06-24T09:01:19.894927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:01:22.743291Z","iopub.execute_input":"2023-06-24T09:01:22.743689Z","iopub.status.idle":"2023-06-24T09:01:22.773132Z","shell.execute_reply.started":"2023-06-24T09:01:22.743657Z","shell.execute_reply":"2023-06-24T09:01:22.772058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. EXPLORATORY DATA ANALYSIS (EDA)","metadata":{}},{"cell_type":"markdown","source":"**2.1 Inspect the 'PATH' Column**","metadata":{}},{"cell_type":"code","source":"np.array(list(df[\"path\"].value_counts().to_dict().values())).min()\ndf[\"path\"].describe().to_frame().T","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:01:26.107288Z","iopub.execute_input":"2023-06-24T09:01:26.107661Z","iopub.status.idle":"2023-06-24T09:01:26.132959Z","shell.execute_reply.started":"2023-06-24T09:01:26.107632Z","shell.execute_reply":"2023-06-24T09:01:26.131923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The path column is simply the path to the landmark file (parquet).\n\nNumber unique paths: 68\n\nMinimum Number of repeated path is: 287\n\nMaximum Number of repeated path is: 1000","metadata":{}},{"cell_type":"markdown","source":"**2.2 Inspect the 'participant_id' Column**","metadata":{}},{"cell_type":"code","source":"df[\"participant_id\"].astype(str).describe().to_frame().T","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:01:29.358587Z","iopub.execute_input":"2023-06-24T09:01:29.35931Z","iopub.status.idle":"2023-06-24T09:01:29.417172Z","shell.execute_reply.started":"2023-06-24T09:01:29.359278Z","shell.execute_reply":"2023-06-24T09:01:29.41612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The participant_id statistics indicate a varied distribution of data contributions among participants, with some participants contributing more examples than others.\n\nNumber of Unique Participants: 94\n\nAverage Number of Rows Per Participant: 715.82\n\nStandard Deviation in Counts Per Participant: 230.86\n\nMinimum Number of Examples For One Participant: 1\n\nMaximum Number of Examples For One Participant: 1537","metadata":{}},{"cell_type":"code","source":"color_scheme = [\"#4f000b\", \"#720026\", \"#ce4257\", \"#ff7f51\", \"#ff9b54\"]\n#The column is set to strings as it is an ID\ndf[\"participant_id\"] = df[\"participant_id\"].astype(str)\n\n# Calculate the counts for each participant_id\ncounts = df[\"participant_id\"].value_counts()\n\n# Set up the figure and axes\nfig, ax = plt.subplots(figsize=(15, 6))\n\n# Plot the histogram\nbars = ax.bar(counts.index, counts.values, color=color_scheme[1])\n\n# Set the labels and title\nax.set_xlabel(\"Participant ID\")\nax.set_ylabel(\"Total Row Count\")\nax.set_title(\"Row Counts by Participant ID\")\n\n# Rotate the x-axis labels if needed\nplt.xticks(rotation=90, ha='center')\nplt.xlim(-1, 94)\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:01:32.734921Z","iopub.execute_input":"2023-06-24T09:01:32.73531Z","iopub.status.idle":"2023-06-24T09:01:33.609257Z","shell.execute_reply.started":"2023-06-24T09:01:32.735277Z","shell.execute_reply":"2023-06-24T09:01:33.608265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**2.3 Inspect the 'sequence_id' Column**","metadata":{}},{"cell_type":"code","source":"df[\"sequence_id\"].astype(str).describe().to_frame().T","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:01:38.117447Z","iopub.execute_input":"2023-06-24T09:01:38.118172Z","iopub.status.idle":"2023-06-24T09:01:38.194382Z","shell.execute_reply.started":"2023-06-24T09:01:38.118138Z","shell.execute_reply":"2023-06-24T09:01:38.193341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A unique identifier for the landmark sequence. Each data file may contain many sequences. Every value is unique for every row","metadata":{}},{"cell_type":"markdown","source":"**2.4 Inspect the 'phrases' Column**","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\n\ndf['phrase_len'] = df.phrase.str.len()\nmetadata['phrase_len'] = metadata.phrase.str.len()\n\nfor param in ['text.color', 'axes.labelcolor', 'xtick.color', 'ytick.color']:\n    plt.rcParams[param] = '#000000'  # very light grey\n\nfor param in ['figure.facecolor', 'axes.facecolor', 'savefig.facecolor']:\n    plt.rcParams[param] = '#ffffff'  # bluish dark grey\n\nfig, axs = plt.subplots(1, 1, figsize=(10, 7), tight_layout=True)\n\n# Remove axes splines\nfor s in ['top', 'bottom', 'left', 'right']:\n    axs.spines[s].set_visible(False)\n\n# Remove x, y ticks\naxs.xaxis.set_ticks_position('none')\naxs.yaxis.set_ticks_position('none')\n\n# Add padding between axes and labels\naxs.xaxis.set_tick_params(pad=5)\naxs.yaxis.set_tick_params(pad=10)\ncolor_scheme = [\"#4f000b\", \"#720026\", \"#ce4257\", \"#ff7f51\", \"#ff9b54\"]\n\n# Add x, y gridlines\naxs.grid(visible=True, color='grey', linestyle='-.', linewidth=0.5, alpha=0.6)\n\n\nplt.subplot(1, 2, 1)\nsns.histplot(df.phrase_len, kde=True, binwidth = 2, color=color_scheme[0])\nplt.title('Character occurences in each phrase in training set')\nplt.xlabel('Phrase length')\nplt.ylabel('Sample Count')\n\nplt.subplot(1, 2, 2)\nsns.histplot(metadata.phrase_len, kde=True, binwidth = 2, color=color_scheme[0])\nplt.title('    and Supplementary metadata')\nplt.xlabel('Unique characters')\nplt.ylabel('Sample Count')\nplt.grid(axis='y')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:01:40.887043Z","iopub.execute_input":"2023-06-24T09:01:40.888051Z","iopub.status.idle":"2023-06-24T09:01:42.278642Z","shell.execute_reply.started":"2023-06-24T09:01:40.888014Z","shell.execute_reply":"2023-06-24T09:01:42.277879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**2.4.1 What phrases are we trying to predict?**","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ntrain = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\nfig, ax = plt.subplots(figsize=(8, 8))\ntrain[\"phrase\"].value_counts().head(50).sort_values(ascending=True).plot(\n    kind=\"barh\", ax=ax, title=\"Top 50 Signs in Training Dataset\"\n)\nax.set_xlabel(\"Number of Training Examples\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:01:47.747229Z","iopub.execute_input":"2023-06-24T09:01:47.7476Z","iopub.status.idle":"2023-06-24T09:01:48.576545Z","shell.execute_reply.started":"2023-06-24T09:01:47.747571Z","shell.execute_reply":"2023-06-24T09:01:48.575477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfig, ax = plt.subplots(figsize=(8, 8))\ntrain[\"phrase\"].value_counts().tail(50).sort_values(ascending=True).plot(\n    kind=\"barh\", ax=ax, title=\"Bottom 50 Signs in Training Dataset\"\n)\nax.set_xlabel(\"Number of Training Examples\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:01:52.415113Z","iopub.execute_input":"2023-06-24T09:01:52.415507Z","iopub.status.idle":"2023-06-24T09:01:53.094433Z","shell.execute_reply.started":"2023-06-24T09:01:52.415476Z","shell.execute_reply":"2023-06-24T09:01:53.093436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**2.5 Inspecting the 'parquet' column** ","metadata":{}},{"cell_type":"code","source":"sample = pd.read_parquet(\"/kaggle/input/asl-fingerspelling/train_landmarks/1019715464.parquet\")","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:01:56.781471Z","iopub.execute_input":"2023-06-24T09:01:56.781879Z","iopub.status.idle":"2023-06-24T09:02:12.805956Z","shell.execute_reply.started":"2023-06-24T09:01:56.781845Z","shell.execute_reply":"2023-06-24T09:02:12.805105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Sample shape = {sample.shape}\")\nsample.sample(10)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:02:15.569864Z","iopub.execute_input":"2023-06-24T09:02:15.570258Z","iopub.status.idle":"2023-06-24T09:02:15.606542Z","shell.execute_reply.started":"2023-06-24T09:02:15.570226Z","shell.execute_reply":"2023-06-24T09:02:15.60558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample.describe()","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:02:19.856898Z","iopub.execute_input":"2023-06-24T09:02:19.857265Z","iopub.status.idle":"2023-06-24T09:02:30.955834Z","shell.execute_reply.started":"2023-06-24T09:02:19.857236Z","shell.execute_reply":"2023-06-24T09:02:30.954764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"MULTI_HAND_LANDMARKS Collection of detected/tracked hands, where each hand is represented as a list of 21 hand landmarks and each landmark is composed of x, y and z. x and y are normalized to [0.0, 1.0] by the image width and height respectively. z represents the landmark depth with the depth at the wrist being the origin, and the smaller the value the closer the landmark is to the camera. The magnitude of z uses roughly the same scale as x.\n\nInaddition, there are few negative coordinates above. negative values are not expected for x and y perhaps. It is also important to note that for a good chunk of the video either or both hands will not be visible or in other words, will not have any landmark data.\n\nLets extract the columns and check the nans in either hands","metadata":{}},{"cell_type":"code","source":"def get_cols(df, words_pos, words_neg=[], ret_names=True):\n    cols = []\n    names = []\n    for col in df.columns:\n        # Check if column name contains all words\n        if all([w in col for w in words_pos]) and all([w not in col for w in words_neg]):\n            cols.append(df[col])  # Append the entire column to the list\n            names.append(col)\n\n    if ret_names:\n        return cols, names\n    else:\n        return cols","metadata":{"execution":{"iopub.status.busy":"2023-06-24T11:23:13.912012Z","iopub.execute_input":"2023-06-24T11:23:13.912439Z","iopub.status.idle":"2023-06-24T11:23:13.918801Z","shell.execute_reply.started":"2023-06-24T11:23:13.912392Z","shell.execute_reply":"2023-06-24T11:23:13.9179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Landmark Indices for Left/Right hand without z axis in raw data\nLH_Index, LEFT_HAND_NAME = get_cols(sample, ['left_hand'], ['z'])\nRH_Index,RIGHT_HAND_NAME = get_cols(sample, ['right_hand'], ['z'])\n#RIGHT_HAND_NAMES0.insert(0, \"frame\")\nLEFT_HAND_NAME.insert(0, \"frame\")\nCOLUMNS = np.concatenate((LEFT_HAND_NAME, RIGHT_HAND_NAME))\n\nN_COLS0 = len(COLUMNS)\n\nprint(f'Total number of columns: {N_COLS0}')","metadata":{"execution":{"iopub.status.busy":"2023-06-24T11:23:16.984609Z","iopub.execute_input":"2023-06-24T11:23:16.984994Z","iopub.status.idle":"2023-06-24T11:23:16.997168Z","shell.execute_reply.started":"2023-06-24T11:23:16.98496Z","shell.execute_reply":"2023-06-24T11:23:16.996252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RIGHT_HAND = sample.loc[:, RIGHT_HAND_NAME] \nLEFT_HAND = sample.loc[:, LEFT_HAND_NAME] \nRIGHT_HAND","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:02:48.161675Z","iopub.execute_input":"2023-06-24T09:02:48.162038Z","iopub.status.idle":"2023-06-24T09:02:48.221768Z","shell.execute_reply.started":"2023-06-24T09:02:48.162009Z","shell.execute_reply":"2023-06-24T09:02:48.220753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Percentage of nulls in Left Hand data = {100*np.mean(LEFT_HAND['x_left_hand_0'].isnull()):.02f} %\")\nprint(f\"Percentage of nulls in Right Hand data = {100*np.mean(RIGHT_HAND['x_right_hand_0'].isnull()):.02f} %\")","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:02:53.862962Z","iopub.execute_input":"2023-06-24T09:02:53.863336Z","iopub.status.idle":"2023-06-24T09:02:53.87134Z","shell.execute_reply.started":"2023-06-24T09:02:53.863307Z","shell.execute_reply":"2023-06-24T09:02:53.870303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Preprocessing Data ","metadata":{}},{"cell_type":"code","source":"sample = pd.read_parquet('/kaggle/input/asl-fingerspelling/train_landmarks/1021040628.parquet')\nLANDMARK_FILES_DIR = \"/kaggle/input/asl-fingerspelling/train_landmarks\"\nTRAIN_FILE = \"/kaggle/input/asl-fingerspelling/train.csv\"\nlabel_map = json.load(open(\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\", \"r\"))","metadata":{"execution":{"iopub.status.busy":"2023-06-24T11:23:21.043528Z","iopub.execute_input":"2023-06-24T11:23:21.044431Z","iopub.status.idle":"2023-06-24T11:23:23.960067Z","shell.execute_reply.started":"2023-06-24T11:23:21.044374Z","shell.execute_reply":"2023-06-24T11:23:23.959263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(label_map)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T11:23:34.304761Z","iopub.execute_input":"2023-06-24T11:23:34.305143Z","iopub.status.idle":"2023-06-24T11:23:34.312342Z","shell.execute_reply.started":"2023-06-24T11:23:34.305112Z","shell.execute_reply":"2023-06-24T11:23:34.311252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So there are 59 characters in total in the vocabulary. :\n\n- alphabets a-z : total 26 characters\n- digits : 0-9 , 10 in total\n- The rest are special characters.","metadata":{}},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    #start_mem = df.memory_usage().sum() / 1024**2\n    #print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n\n    for col in df.columns:\n        col_type = df[col].dtype\n\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-24T11:23:38.158302Z","iopub.execute_input":"2023-06-24T11:23:38.158734Z","iopub.status.idle":"2023-06-24T11:23:38.169617Z","shell.execute_reply.started":"2023-06-24T11:23:38.158702Z","shell.execute_reply":"2023-06-24T11:23:38.168265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import multiprocessing as mp\nimport pandas as pd\nimport torch\nimport numpy as np\n\n# Function to process a single parquet file\ndef process_parquet(row):\n    path = os.path.join(\"/kaggle/input/asl-fingerspelling\", row[1].path)\n    data_columns = COLUMNS\n    landmark_df = pd.read_parquet(path, columns=data_columns)\n\n    # Group the landmarks by sequence_id\n    grouped_landmarks = landmark_df.groupby('sequence_id')\n\n    # Initialize empty lists to store features and labels\n    features = []\n    labels = []\n\n    # Iterate over each sequence_id\n    for sequence_id, group in grouped_landmarks:\n        # Get the label for the sequence\n        phrase = df.loc[df['sequence_id'] == sequence_id, 'phrase'].iloc[0]\n\n        # Map each letter in the phrase using label_map\n        mapped_phrase = [letter for letter in phrase]\n        # Create a new Series with sequence and mapped_phrase\n        result_series = pd.DataFrame({'sequence_id': sequence_id, 'mapped_phrase': mapped_phrase}) \n        result_series['label'] = result_series['mapped_phrase'].map(label_map).astype(np.int8)  \n\n        # Convert the label Series to a list\n        #label_list = result_series['label'].tolist()\n\n        # Initialize an empty feature vector for the sequence\n        sequence_features = []\n        \n        # Iterate over each landmark index\n        for landmark_index in range(20):\n            # Generate feature names for x, y, z coordinates\n            x_feature = f'x_right_hand_{landmark_index}'\n            y_feature = f'y_right_hand_{landmark_index}'\n\n            # Get the x, y, z coordinates for the landmark\n            x = group[x_feature].values.astype(np.float16)\n            y = group[y_feature].values.astype(np.float16)\n\n            # Replace NaN values with 0\n            x[np.isnan(x)] = 0.0\n            y[np.isnan(y)] = 0.0\n                        \n            # Perform feature transformations or calculations\n            x = torch.tensor(x).contiguous().view(-1, x.shape[0])\n            y = torch.tensor(y).contiguous().view(-1, y.shape[0])\n            \n            x_mean = torch.mean(x, 0) \n            y_mean = torch.mean(y, 0) \n\n            x_std = torch.std(x, 1) \n            y_std = torch.std(y, 1) \n\n            # Add the calculated features to the sequence feature vector\n            sequence_features = torch.cat([x_mean, y_mean,x_std,y_std], axis=0)\n            #sequence_features = torch.where(torch.isnan(sequence_features), torch.tensor(0.0, dtype=torch.float32), sequence_features)\n\n            diff = 3258 - sequence_features.shape[0]\n            if diff > 0:\n                padding = torch.zeros(diff)\n                sequence_features = torch.cat((sequence_features, padding))\n            \n            features = sequence_features[:3258].cpu().numpy()\n            return features, result_series['label']\n\nif __name__ == '__main__':\n    df = pd.read_csv(TRAIN_FILE)\n    df = reduce_mem_usage(df) \n    df2 = df.head(30)\n\n    max_label_length = 30\n    all_features = np.zeros((df.shape[0], 3258))\n    labels = np.zeros((df.shape[0], max_label_length))\n    #all_features = []\n    #labels = []\n\n    # Process parquet files in parallel\n    with mp.Pool() as pool:\n        results = pool.imap(process_parquet, df.iterrows(), chunksize=250)\n        for i, (x, y) in tqdm(enumerate(results), total=df.shape[0]):\n            all_features[i, :] = x\n            labels[i, :len(y)] = y.values.reshape(1, -1)\n\n    # Print the shapes of the tensors\n    np.save(\"feature_data.npy\", all_features)\n    np.save(\"feature_labels.npy\", labels)\n\n    print(\"Features tensor shape:\", all_features.shape)\n    print(\"Labels tensor shape:\", labels.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T11:23:42.843881Z","iopub.execute_input":"2023-06-24T11:23:42.844271Z","iopub.status.idle":"2023-06-24T11:24:32.030892Z","shell.execute_reply.started":"2023-06-24T11:23:42.844238Z","shell.execute_reply":"2023-06-24T11:24:32.029487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Evaluation Metric","metadata":{}},{"cell_type":"markdown","source":"The Levenshtein distance, also known as the edit distance, quantifies the dissimilarity between two strings by measuring the minimum number of single-character edits (insertions, deletions, or substitutions) required to transform one string into another. This metric provides a valuable measure of how well the predicted ASL sequence matches the ground truth or reference sequence.\n\nTo calculate the Levenshtein distance for ASL recognition, the predicted ASL sequence and the reference or ground truth sequence are compared character by character. Each character is treated as a token, representing a specific sign or gesture. The Levenshtein distance is then computed by determining the minimum number of edit operations needed to transform the predicted sequence into the reference sequence or vice versa.\n\nThe evaluation metric for this contest is the normalized total Levenshtein distance. The formula for calculating the metric is as follows:\n\nMetric = (N - D) / N\n\nWhere:\n\nN is the total number of characters in the labels.\nD is the total Levenshtein distance.\n\nTo calculate the metric, you would need the labels data and the predicted sequence. The Levenshtein distance measures the minimum number of single-character edits (insertions, deletions, or substitutions) required to change one sequence into another.\n\nIn the given context, it seems that the labels are provided in the \"phrase\" column of the [train/supplemental_metadata].csv file. The predicted sequence can be obtained from the landmark data files in the [train/supplemental]_landmarks/ directory.\n\nTo calculate the total Levenshtein distance, you would need to compare each character in the labels with the corresponding character in the predicted sequence and count the number of edits required.\n\nFinally, you can plug the values of N (total characters in the labels) and D (total Levenshtein distance) into the formula to compute the metric. The resulting value will give you an indication of the accuracy of the predicted sequence compared to the labels, with higher values indicating better performance.\n\nNote: Since the specific implementation details are not provided, you would need to write code or use existing libraries to calculate the Levenshtein distance and implement the metric calculation\n\nThe LD is explained in depth in this discussion","metadata":{}},{"cell_type":"code","source":"from Levenshtein import distance\n#Using the dynamic programming approach for calculating the Levenshtein distance\n\ndef levenshteinDistanceDP(token1, token2):\n    # Create a 2-D matrix \n    distances = np.zeros((len(token1) + 1, len(token2) + 1))\n    \n    #Initialize the first row and column, Row index is fixed to 0 and the variable t1 is used to define the column index. \n    for t1 in range(len(token1) + 1):\n        distances[t1][0] = t1\n        \n    #Column index of the distances array is now fixed to 0, while the loop variable t2 is used to define the index of the rows\n    for t2 in range(len(token2) + 1):\n        distances[0][t2] = t2\n    a = 0\n    b = 0\n    c = 0\n    \n    #Inside the loops the distances are calculated for all combinations of prefixes from the two words. \n    for t1 in range(1, len(token1) + 1):\n        for t2 in range(1, len(token2) + 1):\n            if (token1[t1-1] == token2[t2-1]):\n                distances[t1][t2] = distances[t1 - 1][t2 - 1]\n                \n            #If the two characters are not equal, then the distance in the current cell is equal to the\n            #minimum of the three existing values in the 2 x 2 matrix after adding a cost of 1\n            else:\n                a = distances[t1][t2 - 1]\n                b = distances[t1 - 1][t2]\n                c = distances[t1 - 1][t2 - 1]\n                \n                if (a <= b and a <= c):\n                    distances[t1][t2] = a + 1\n                elif (b <= a and b <= c):\n                    distances[t1][t2] = b + 1\n                else:\n                    distances[t1][t2] = c + 1\n                    \n    #Print its contents \n    printDistances(distances, len(token1), len(token2))\n    \n    #returning the calculated distance between the two words\n    return distances[len(token1)][len(token2)]\ndef printDistances(distances, token1Length, token2Length):\n    for t1 in range(token1Length + 1):\n        for t2 in range(token2Length + 1):\n            print(int(distances[t1][t2]), end=\" \")\n        print()\n        \nphase1 = '3 creekhouse'\nphase2 = 'scales/kuhaylah'\n\n#Calling levenshteinDistanceDP function, \n#It returns an integer representing the distance between them\n\nprint(\"Printing The Distance Matrix:\")\nprint(f\" \\nThe Levenshtein distance of phrases = {levenshteinDistanceDP(phase1, phase2):.02f} \")","metadata":{"execution":{"iopub.status.busy":"2023-06-24T11:22:48.592344Z","iopub.execute_input":"2023-06-24T11:22:48.592859Z","iopub.status.idle":"2023-06-24T11:22:48.712658Z","shell.execute_reply.started":"2023-06-24T11:22:48.592821Z","shell.execute_reply":"2023-06-24T11:22:48.711505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. VISUALIZATION ","metadata":{}},{"cell_type":"code","source":"import pandas as pd,numpy as np,os\nimport json\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.io as pio\nfrom pathlib import Path\n\nfrom sklearn.decomposition import PCA\nfrom sklearn.cluster import KMeans\nfrom matplotlib.colors import ListedColormap\nfrom sklearn.preprocessing import normalize\n\nprint(\"importing\")","metadata":{"execution":{"iopub.status.busy":"2023-06-24T08:48:27.766987Z","iopub.execute_input":"2023-06-24T08:48:27.767381Z","iopub.status.idle":"2023-06-24T08:48:29.641544Z","shell.execute_reply.started":"2023-06-24T08:48:27.76735Z","shell.execute_reply":"2023-06-24T08:48:29.640812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LANDMARK_FILES_DIR = \"/kaggle/input/asl-fingerspelling/train_landmarks\"\nTRAIN_FILE = \"/kaggle/input/asl-fingerspelling/train.csv\"\nlabel_map = json.load(open(\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\", \"r\"))","metadata":{"execution":{"iopub.status.busy":"2023-06-24T08:48:37.111507Z","iopub.execute_input":"2023-06-24T08:48:37.111858Z","iopub.status.idle":"2023-06-24T08:48:37.116909Z","shell.execute_reply.started":"2023-06-24T08:48:37.111828Z","shell.execute_reply":"2023-06-24T08:48:37.115934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\n\nimport json\nfrom tqdm import tqdm\nimport numpy as np\nimport pandas as pd\n\nimport torch\nimport torch.nn as nn\n\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\nimport warnings\nwarnings.filterwarnings(action='ignore')","metadata":{"execution":{"iopub.status.busy":"2023-06-24T08:54:41.304824Z","iopub.execute_input":"2023-06-24T08:54:41.305299Z","iopub.status.idle":"2023-06-24T08:54:41.31334Z","shell.execute_reply.started":"2023-06-24T08:54:41.305254Z","shell.execute_reply":"2023-06-24T08:54:41.312154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LANDMARK_FILES_DIR = \"/kaggle/input/asl-fingerspelling/train_landmarks\"\nTRAIN_FILE = \"/kaggle/input/asl-fingerspelling/train.csv\"\nlabel_map = json.load(open(\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\", \"r\"))","metadata":{"execution":{"iopub.status.busy":"2023-06-24T08:49:08.293925Z","iopub.execute_input":"2023-06-24T08:49:08.29461Z","iopub.status.idle":"2023-06-24T08:49:08.300307Z","shell.execute_reply.started":"2023-06-24T08:49:08.294578Z","shell.execute_reply":"2023-06-24T08:49:08.299097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    #start_mem = df.memory_usage().sum() / 1024**2\n    #print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n\n    for col in df.columns:\n        col_type = df[col].dtype\n\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    #end_mem = df.memory_usage().sum() / 1024**2\n    #print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    #print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-24T08:54:45.472642Z","iopub.execute_input":"2023-06-24T08:54:45.473826Z","iopub.status.idle":"2023-06-24T08:54:45.485387Z","shell.execute_reply.started":"2023-06-24T08:54:45.473768Z","shell.execute_reply":"2023-06-24T08:54:45.484569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import multiprocessing as mp\nimport torch.nn.functional as F\nimport pandas as pd\nimport torch\ndef process_parquet(row):\n    path = os.path.join(\"/kaggle/input/asl-fingerspelling\", row[1].path)\n    COLUMNS = np.concatenate((LEFT_HAND_NAME, RIGHT_HAND_NAME))\n    data_columns = COLUMNS\n    landmark_df = pd.read_parquet(path, columns=data_columns)\n    grouped_landmarks = landmark_df.groupby('sequence_id')\n    features = []\n    labels = []\n\n    for sequence_id, group in grouped_landmarks:\n        phrase = df.loc[df['sequence_id'] == sequence_id, 'phrase'].iloc[0]\n        mapped_phrase = [letter for letter in phrase]\n        result_series = pd.DataFrame({'sequence_id': sequence_id, 'mapped_phrase': mapped_phrase}) \n        result_series['label'] = result_series['mapped_phrase'].map(label_map).astype(np.int8)  \n        \n        label_list = result_series['label'].tolist()\n        sequence_features = []\n        for landmark_index in range(20):\n            x_feature = f'x_right_hand_{landmark_index}'\n            y_feature = f'y_right_hand_{landmark_index}'\n            x = group[x_feature].values.astype(np.float16)\n            y = group[y_feature].values.astype(np.float16)\n            x = torch.tensor(x).contiguous().view(-1, x.shape[0])\n            y = torch.tensor(y).contiguous().view(-1, y.shape[0])\n\n            x = x[:,~torch.any(torch.isnan(x), dim=0)]\n            y = y[:,~torch.any(torch.isnan(y), dim=0)]\n            \n            x_mean = torch.mean(x, 0) \n            y_mean = torch.mean(y, 0) \n\n            x_std = torch.std(x,1) \n            y_std = torch.std(y,1) \n\n            sequence_features = torch.cat([x_mean,y_mean,x_std,y_std], axis=0)\n            sequence_features = torch.where(torch.isnan(sequence_features), torch.tensor(0.0, dtype=torch.float32), sequence_features)\n\n            diff = 3258 - sequence_features.shape[0]\n            if (diff >= 0):\n                padding = torch.zeros(diff)\n                sequence_features = torch.cat((sequence_features, padding))\n            features =  sequence_features[:3258].cpu().numpy()\n            return features,result_series['label']\n\nall_features = []\nall_labels = []\n\ndf = pd.read_csv(TRAIN_FILE)\ndf = reduce_mem_usage(df) \ndf2 = df.head(30)\n\nmax_label_length = 30\nall_features = np.zeros((df.shape[0], 3258))\nlabels = np.zeros((df.shape[0], max_label_length))\n    \n# Process parquet files in parallel\nwith mp.Pool() as pool:\n    results = pool.imap(process_parquet, df.iterrows(),  chunksize=250)\n    for i, (x,y) in tqdm(enumerate(results), total=df.shape[0]):\n        all_features[i,:] = x\n        labels[i,:len(y)] = y.values.reshape(1, -1)\n\nnp.save(\"feature_data.npy\", all_features)\nnp.save(\"feature_labels.npy\", labels)\n\nprint(\"Features tensor shape:\", all_features.shape)\nprint(\"Labels tensor shape:\", labels.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T09:05:49.693764Z","iopub.execute_input":"2023-06-24T09:05:49.694684Z","iopub.status.idle":"2023-06-24T11:02:29.244953Z","shell.execute_reply.started":"2023-06-24T09:05:49.694646Z","shell.execute_reply":"2023-06-24T11:02:29.243382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. Training","metadata":{}},{"cell_type":"markdown","source":"I trained with neural network model using PyTorch. The algorithm used in this code is not explicitly mentioned, but based on the code structure and components, it appears to be a classification task using a neural network with the Adam optimizer and CrossEntropyLoss as the loss function.","metadata":{}},{"cell_type":"code","source":"datay = np.load(\"/kaggle/input/feature-data-labels/feature_labels.npy\")\ndatax = np.load(\"/kaggle/input/feature-data-labels/feature_data.npy\")","metadata":{"execution":{"iopub.status.busy":"2023-06-24T11:29:28.891305Z","iopub.execute_input":"2023-06-24T11:29:28.891881Z","iopub.status.idle":"2023-06-24T11:29:40.226155Z","shell.execute_reply.started":"2023-06-24T11:29:28.891835Z","shell.execute_reply":"2023-06-24T11:29:40.225018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datax.shape[1]","metadata":{"execution":{"iopub.status.busy":"2023-06-24T11:30:01.192584Z","iopub.execute_input":"2023-06-24T11:30:01.193016Z","iopub.status.idle":"2023-06-24T11:30:01.201629Z","shell.execute_reply.started":"2023-06-24T11:30:01.192986Z","shell.execute_reply":"2023-06-24T11:30:01.200229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ASLModel(nn.Module):\n    def __init__(self, p):\n        super(ASLModel, self).__init__()\n        self.dropout = nn.Dropout(p)\n        self.layer0 = nn.Linear(3258, 1024)\n        self.layer1 = nn.Linear(1024, 512)\n        self.layer2 = nn.Linear(512, 30)\n\n        \n    def forward(self, x):\n        x = self.layer0(x)\n        x = self.dropout(x)\n        x = self.layer1(x)\n        x = self.layer2(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-06-24T11:30:14.654961Z","iopub.execute_input":"2023-06-24T11:30:14.655374Z","iopub.status.idle":"2023-06-24T11:30:14.663436Z","shell.execute_reply.started":"2023-06-24T11:30:14.655343Z","shell.execute_reply":"2023-06-24T11:30:14.662064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import Levenshtein\n\ndef calculate_normalized_levenshtein_distance(pred_strings, target_strings):\n    total_distance = 0\n    total_length = 0\n\n    for pred, target in zip(pred_strings, target_strings):\n        distance = Levenshtein.distance(pred, target)\n        total_distance += distance\n        total_length += len(target)\n\n    normalized_distance = total_distance / total_length\n    return normalized_distance","metadata":{"execution":{"iopub.status.busy":"2023-06-24T11:30:27.424185Z","iopub.execute_input":"2023-06-24T11:30:27.424638Z","iopub.status.idle":"2023-06-24T11:30:27.43124Z","shell.execute_reply.started":"2023-06-24T11:30:27.424601Z","shell.execute_reply":"2023-06-24T11:30:27.429911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MultiLabelBinarizer\nimport Levenshtein\n\ndef calculate_levenshtein_distance(pred_labels, target_labels):\n    distance = 0\n    for pred, target in zip(pred_labels, target_labels):\n        distance += Levenshtein.distance(pred, target)\n    return distance\n\nclass ASLDataset(Dataset):\n    def __init__(self, datax, datay):\n        self.datax = datax\n        self.datay = datay\n        \n    def __getitem__(self, index):\n        return self.datax[index,:], self.datay[index]\n        \n    def __len__(self):\n        return len(self.datay)\n    \n# Data Split\ntrainx, testx, trainy, testy = train_test_split(datax, datay, test_size=0.15, random_state=42)\n\n# Convert data to PyTorch tensors\ntrainx = torch.from_numpy(trainx).float()\ntrainy = torch.from_numpy(trainy).float()\ntestx = torch.from_numpy(testx).float()\ntesty = torch.from_numpy(testy).float()\n\n# Set device\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Data Preparation\ntrain_data = ASLDataset(trainx, trainy)\ntest_data = ASLDataset(testx, testy)\n\n# DataLoader\nBATCH_SIZE = 128\ntrain_loader = DataLoader(train_data, batch_size=BATCH_SIZE, shuffle=True)\ntest_loader = DataLoader(test_data, batch_size=BATCH_SIZE, shuffle=False)\n\n# Model Definition\nclass ASLModel(nn.Module):\n    def __init__(self, input_size, output_size):\n        super(ASLModel, self).__init__()\n        self.fc1 = nn.Linear(input_size, 2048)\n        self.fc2 = nn.Linear(2048, 1024)\n        self.fc3 = nn.Linear(1024, output_size)\n        #self.dropout = nn.Dropout(0.2)  \n\n    def forward(self, x):\n        x = torch.relu(self.fc1(x))\n        #x = self.dropout(x)\n        x = torch.relu(self.fc2(x))\n        #x = self.dropout(x)\n        x = self.fc3(x)\n        return x\n\n# Model Initialization\nmodel = ASLModel(input_size=trainx.shape[1], output_size=trainy.shape[1]).to(device)\n\n# Optimization Setup\ncriterion = nn.MSELoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=0.001)\n#optimizer = torch.optim.SGD(model.parameters(), lr=0.001)  \n#optimizer = torch.optim.AdamW(model.parameters(), lr=0.005)\n\n# Training Loop\nEPOCHS = 50\nfor epoch in range(EPOCHS):\n    model.train()\n    train_loss = 0.0\n    train_correct = 0\n\n    for inputs, targets in train_loader:\n        inputs = inputs.to(device)\n        targets = targets.to(device)\n        #print('targets = ',targets)\n\n        optimizer.zero_grad()\n        outputs = model(inputs)\n        #print('models output = ',outputs)\n        loss = criterion(outputs, targets)\n        loss.backward()\n        optimizer.step()\n\n        train_loss += loss.item()\n        train_correct += (outputs.round() == targets).sum().item()\n\n    train_loss /= len(train_loader)\n    train_accuracy = train_correct / len(train_data)\n\n    # Evaluation\n    model.eval()\n    test_loss = 0.0\n    test_correct = 0\n    levenshtein_distance = 0\n\n    with torch.no_grad():\n        for inputs, targets in test_loader:\n            inputs = inputs.to(device)\n            targets = targets.to(device)\n\n            outputs = model(inputs)\n            loss = criterion(outputs, targets)\n            #print(loss)\n\n            test_loss += loss.item()\n            test_correct += (outputs.round() == targets).sum().item()\n\n            # Define the reverse mapping dictionary\n            reverse_label_map = {v: k for k, v in label_map.items()}\n            outputs_array = outputs.detach().cpu().numpy()\n            targets_array = targets.detach().cpu().numpy()\n\n            # Convert predictions and targets to letter sequences\n            pred_labels = [[reverse_label_map[label] for label in output.nonzero()[0].tolist()] if len(output.nonzero()[0]) > 0 else [] for output in outputs_array.round()]\n            target_labels = [[reverse_label_map[label] for label in target.nonzero()[0].tolist()] if len(target.nonzero()[0]) > 0 else [] for target in targets_array.round()]\n            # Calculate Levenshtein distance\n            #print(pred_labels)\n            levenshtein_distance += calculate_levenshtein_distance(pred_labels, target_labels)\n\n    test_loss /= len(test_loader)\n    test_accuracy = test_correct / len(test_data)\n    average_levenshtein_distance = levenshtein_distance / len(test_data)\n\n    # Print epoch results\n    print(f\"Epoch {epoch+1}/{EPOCHS}\")\n    print(f\"Train Loss: {train_loss:.4f} | Train Accuracy: {train_accuracy:.4f}\")\n    print(f\"Test Loss: {test_loss:.4f} | Test Accuracy: {test_accuracy:.4f}\")\n    print(f\"Average Levenshtein Distance: {average_levenshtein_distance:.4f}\")\n    print(\"=\" * 50)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T11:32:03.61554Z","iopub.execute_input":"2023-06-24T11:32:03.615994Z","iopub.status.idle":"2023-06-24T12:12:53.204006Z","shell.execute_reply.started":"2023-06-24T11:32:03.615963Z","shell.execute_reply":"2023-06-24T12:12:53.202205Z"},"trusted":true},"execution_count":null,"outputs":[]}]}