{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-09T11:30:15.809867Z","iopub.execute_input":"2023-06-09T11:30:15.810235Z","iopub.status.idle":"2023-06-09T11:30:15.849809Z","shell.execute_reply.started":"2023-06-09T11:30:15.810209Z","shell.execute_reply":"2023-06-09T11:30:15.848607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nfrom matplotlib.animation import FuncAnimation\nfrom IPython.display import HTML","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:07:55.336303Z","iopub.execute_input":"2023-06-09T12:07:55.336756Z","iopub.status.idle":"2023-06-09T12:07:55.388661Z","shell.execute_reply.started":"2023-06-09T12:07:55.33672Z","shell.execute_reply":"2023-06-09T12:07:55.387314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\ndf","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:09:05.781748Z","iopub.execute_input":"2023-06-09T12:09:05.782257Z","iopub.status.idle":"2023-06-09T12:09:05.963479Z","shell.execute_reply.started":"2023-06-09T12:09:05.78222Z","shell.execute_reply":"2023-06-09T12:09:05.962285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata = pd.read_csv(\"/kaggle/input/asl-fingerspelling/supplemental_metadata.csv\")\nmetadata","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:08:45.191692Z","iopub.execute_input":"2023-06-09T12:08:45.192119Z","iopub.status.idle":"2023-06-09T12:08:45.283022Z","shell.execute_reply.started":"2023-06-09T12:08:45.192089Z","shell.execute_reply":"2023-06-09T12:08:45.281766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'no of files in train daata {df.shape[0]}')\nprint(f'no of unique phrase in training daata {df.phrase.nunique()}')\nprint(f'no of participants  {df.participant_id.nunique()}')\nprint(f'no of participants  {df.participant_id.nunique()}')\nprint(f'no of unique sequences {df.sequence_id.nunique()}')\n","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:13:48.993052Z","iopub.execute_input":"2023-06-09T12:13:48.993569Z","iopub.status.idle":"2023-06-09T12:13:49.027314Z","shell.execute_reply.started":"2023-06-09T12:13:48.993534Z","shell.execute_reply":"2023-06-09T12:13:49.026082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Total files in metadata: {metadata.shape[0]}\")\nprint(f\"Total Signers in metadata : {(metadata.participant_id.nunique())}\")\nprint(f\"Total unique phrases in metadata : {metadata.phrase.nunique()}\")","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:19:16.615985Z","iopub.execute_input":"2023-06-09T12:19:16.616453Z","iopub.status.idle":"2023-06-09T12:19:16.631077Z","shell.execute_reply.started":"2023-06-09T12:19:16.616418Z","shell.execute_reply":"2023-06-09T12:19:16.629766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ###Some phrases are repeated in train data (metadata)","metadata":{}},{"cell_type":"code","source":"train_unique = df.participant_id.unique()\nmeta_unique = metadata.participant_id.unique()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:17:15.049774Z","iopub.execute_input":"2023-06-09T12:17:15.050271Z","iopub.status.idle":"2023-06-09T12:17:15.056938Z","shell.execute_reply.started":"2023-06-09T12:17:15.050237Z","shell.execute_reply":"2023-06-09T12:17:15.055707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_unique","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:18:28.420182Z","iopub.execute_input":"2023-06-09T12:18:28.420627Z","iopub.status.idle":"2023-06-09T12:18:28.431614Z","shell.execute_reply.started":"2023-06-09T12:18:28.420594Z","shell.execute_reply":"2023-06-09T12:18:28.430029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"both = []\nfor x in train_unique:\n    if x in meta_unique:\n        both.append(x)\n\nprint(f\"Participants present in both metadata and training set :{both}\\nTotal {len(both)} participants overlapped\")","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:17:54.955114Z","iopub.execute_input":"2023-06-09T12:17:54.955532Z","iopub.status.idle":"2023-06-09T12:17:54.964136Z","shell.execute_reply.started":"2023-06-09T12:17:54.9555Z","shell.execute_reply":"2023-06-09T12:17:54.962528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.phrase.iloc[:20].tolist()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:19:44.920396Z","iopub.execute_input":"2023-06-09T12:19:44.920819Z","iopub.status.idle":"2023-06-09T12:19:44.930719Z","shell.execute_reply.started":"2023-06-09T12:19:44.920788Z","shell.execute_reply":"2023-06-09T12:19:44.929255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata.phrase.iloc[:20].tolist()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:20:20.742665Z","iopub.execute_input":"2023-06-09T12:20:20.743074Z","iopub.status.idle":"2023-06-09T12:20:20.751794Z","shell.execute_reply.started":"2023-06-09T12:20:20.743043Z","shell.execute_reply":"2023-06-09T12:20:20.750403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['phrase_len'] = df.phrase.str.len()\nmetadata['phrase_len'] = metadata.phrase.str.len()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:21:23.812854Z","iopub.execute_input":"2023-06-09T12:21:23.813363Z","iopub.status.idle":"2023-06-09T12:21:23.912745Z","shell.execute_reply.started":"2023-06-09T12:21:23.813307Z","shell.execute_reply":"2023-06-09T12:21:23.911063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8,4))\nplt.title('Character occurences in each phrase in training set')\nplt.hist(df.phrase_len)\nplt.xlabel('Phrase length')\nplt.ylabel('Sample Count')\nplt.grid(axis='y')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:23:15.821945Z","iopub.execute_input":"2023-06-09T12:23:15.822452Z","iopub.status.idle":"2023-06-09T12:23:16.206974Z","shell.execute_reply.started":"2023-06-09T12:23:15.82242Z","shell.execute_reply":"2023-06-09T12:23:16.205445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\nplt.title('Character occurences in each phrase in Supplementary metadata')\nplt.hist(metadata.phrase_len)\nplt.xlabel('Unique characters')\nplt.ylabel('Sample Count')\nplt.grid(axis='y')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:23:43.423286Z","iopub.execute_input":"2023-06-09T12:23:43.423977Z","iopub.status.idle":"2023-06-09T12:23:43.776997Z","shell.execute_reply.started":"2023-06-09T12:23:43.423933Z","shell.execute_reply":"2023-06-09T12:23:43.775468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Load the character_to_prediction json file\nimport json\nchars = json.load(open(\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\"))\nchars","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:24:21.867397Z","iopub.execute_input":"2023-06-09T12:24:21.86779Z","iopub.status.idle":"2023-06-09T12:24:21.881066Z","shell.execute_reply.started":"2023-06-09T12:24:21.867762Z","shell.execute_reply":"2023-06-09T12:24:21.880149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"alphabets = \"abcdefghijklmnopqrstuvwxyz\"\ndigits = \"0123456789\"","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:55:47.152061Z","iopub.execute_input":"2023-06-09T12:55:47.153284Z","iopub.status.idle":"2023-06-09T12:55:47.159342Z","shell.execute_reply.started":"2023-06-09T12:55:47.153239Z","shell.execute_reply":"2023-06-09T12:55:47.157771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"special = []\nfor char in chars.keys():\n    if char not in alphabets and char not in digits:\n        special.append(char)\nprint(\"Special Characters :\")\nspecial","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:55:47.165754Z","iopub.execute_input":"2023-06-09T12:55:47.166206Z","iopub.status.idle":"2023-06-09T12:55:47.179823Z","shell.execute_reply.started":"2023-06-09T12:55:47.166175Z","shell.execute_reply":"2023-06-09T12:55:47.178816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab = {}\nfor sentences in tqdm(df.phrase.tolist()):\n    for char in sentences:\n        try:\n            vocab[char]+=1\n        except:\n            vocab[char] = 1\nprint(f\"Total number of characters present in the training dataset: {len(vocab)}\")\nsorted(vocab.items(),key = lambda x:x[1])\n\n    \n","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:57:11.12062Z","iopub.execute_input":"2023-06-09T12:57:11.121115Z","iopub.status.idle":"2023-06-09T12:57:11.473957Z","shell.execute_reply.started":"2023-06-09T12:57:11.121083Z","shell.execute_reply":"2023-06-09T12:57:11.472572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sorted_vocab = sorted(vocab.items(),key = lambda x:x[1],reverse = True)\nsorted_vocab = dict(sorted_vocab)","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:58:20.873185Z","iopub.execute_input":"2023-06-09T12:58:20.874521Z","iopub.status.idle":"2023-06-09T12:58:20.880575Z","shell.execute_reply.started":"2023-06-09T12:58:20.874478Z","shell.execute_reply":"2023-06-09T12:58:20.879192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\nplt.title('Character Occurance')\nplt.bar(sorted_vocab.keys(),sorted_vocab.values())\nplt.xlabel('Unique characters')\nplt.ylabel('Sample Count')\nplt.grid(axis='y')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:58:45.759968Z","iopub.execute_input":"2023-06-09T12:58:45.760408Z","iopub.status.idle":"2023-06-09T12:58:46.459849Z","shell.execute_reply.started":"2023-06-09T12:58:45.760376Z","shell.execute_reply":"2023-06-09T12:58:46.458196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_path = \"/kaggle/input/asl-fingerspelling\"\nsequence_no = 0\nfile_name = df.path.iloc[sequence_no]\nsequence_id = df.sequence_id.iloc[sequence_no]\nrandom_df = pd.read_parquet(f\"{base_path}/{file_name}\")","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:59:31.386024Z","iopub.execute_input":"2023-06-09T12:59:31.386557Z","iopub.status.idle":"2023-06-09T12:59:48.511009Z","shell.execute_reply.started":"2023-06-09T12:59:31.386517Z","shell.execute_reply":"2023-06-09T12:59:48.508483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_df","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:59:58.007478Z","iopub.execute_input":"2023-06-09T12:59:58.007933Z","iopub.status.idle":"2023-06-09T12:59:58.073603Z","shell.execute_reply.started":"2023-06-09T12:59:58.007891Z","shell.execute_reply":"2023-06-09T12:59:58.072191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_df[~random_df.isnull().any(axis=1)]","metadata":{"execution":{"iopub.status.busy":"2023-06-09T13:01:18.967164Z","iopub.execute_input":"2023-06-09T13:01:18.967818Z","iopub.status.idle":"2023-06-09T13:01:19.576353Z","shell.execute_reply.started":"2023-06-09T13:01:18.967773Z","shell.execute_reply":"2023-06-09T13:01:19.574892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for sequence,sequence_data in random_df.groupby(\"sequence_id\"):\n    if sequence==sequence_id:\n        print(f\"Total Frames for this file {sequence_data.frame.nunique()}\")\n        print(f\"Total Frames without any NaN values for this file {sequence_data[~sequence_data.isnull().any(axis=1)].frame.nunique()}\")","metadata":{"execution":{"iopub.status.busy":"2023-06-09T13:02:22.169662Z","iopub.execute_input":"2023-06-09T13:02:22.170114Z","iopub.status.idle":"2023-06-09T13:02:22.791778Z","shell.execute_reply.started":"2023-06-09T13:02:22.170083Z","shell.execute_reply":"2023-06-09T13:02:22.790255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nnumber_of_frames = {}\n\nfiles_path = \"/kaggle/input/asl-fingerspelling/train_landmarks\"\nall_files = os.listdir(\"/kaggle/input/asl-fingerspelling/train_landmarks\")\nfor file in tqdm(all_files):\n    parq = pd.read_parquet(f\"{files_path}/{file}\")\n    for seq,seq_data in parq.groupby(\"sequence_id\"):\n        total_frames = seq_data.frame.nunique()\n        \n        try:\n            number_of_frames[total_frames]+=1\n        except:\n            number_of_frames[total_frames]=1","metadata":{"execution":{"iopub.status.busy":"2023-06-09T13:02:50.497512Z","iopub.execute_input":"2023-06-09T13:02:50.498049Z","iopub.status.idle":"2023-06-09T13:26:02.396065Z","shell.execute_reply.started":"2023-06-09T13:02:50.498012Z","shell.execute_reply":"2023-06-09T13:26:02.393349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.xlabel(\"Number of frames\")\nplt.ylabel(\"occurances\")\nplt.hist(number_of_frames,bins=[i for i in range(0,1000,100)])","metadata":{},"execution_count":null,"outputs":[]}]}