{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":52950,"databundleVersionId":5973250}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## ✋🏽 Update: Landmarks Normalization\n\n### Changes in this version:\n- **Integrate: Landmarks position normalization**\n- (https://www.kaggle.com/code/lejuin/fingerspelling-data-prep#Center-hand)\n\n## 📈 Update: W&B\n\n### Changes in this version:\n- **Integrate: Metrics monitoring in W&B team workspace**\n- (https://wandb.ai/inaki-rodriguez-reyes-upc-universidad-peruana-de-ciencia/asl-fingerspelling-previews)\n\n## 🔡 Update: Words\n\n### Changes in this version:\n- **Integrate: Filter only words with a-z**\n- (https://www.kaggle.com/code/pauerv/notebook136616d653-v12-overfit)\n- **Reduced dataset size**\n  - Train: 6564\n  - Val: 863\n  - Test: 920","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2026-02-24T20:15:07.77789Z","iopub.execute_input":"2026-02-24T20:15:07.778235Z","iopub.status.idle":"2026-02-24T20:15:10.231595Z","shell.execute_reply.started":"2026-02-24T20:15:07.77821Z","shell.execute_reply":"2026-02-24T20:15:10.230955Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install and configure W&B on this notebook\n!pip install wandb -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:15:14.455559Z","iopub.execute_input":"2026-02-24T20:15:14.455952Z","iopub.status.idle":"2026-02-24T20:15:20.330364Z","shell.execute_reply.started":"2026-02-24T20:15:14.455928Z","shell.execute_reply":"2026-02-24T20:15:20.329599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom kaggle_secrets import UserSecretsClient\nimport wandb\n\nuser_secrets = UserSecretsClient()\nsecret_value = user_secrets.get_secret(\"wandb\")\nos.environ[\"WANDB_API_KEY\"] = secret_value\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:15:24.560279Z","iopub.execute_input":"2026-02-24T20:15:24.560946Z","iopub.status.idle":"2026-02-24T20:15:28.014892Z","shell.execute_reply.started":"2026-02-24T20:15:24.560911Z","shell.execute_reply":"2026-02-24T20:15:28.014332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport json\n\n# Lectura de parquets \nimport pyarrow.parquet as pq\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\n\nSEED = 42\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\n\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:15:30.241093Z","iopub.execute_input":"2026-02-24T20:15:30.241375Z","iopub.status.idle":"2026-02-24T20:15:36.550242Z","shell.execute_reply.started":"2026-02-24T20:15:30.241353Z","shell.execute_reply":"2026-02-24T20:15:36.549456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_CSV_PATH = \"/kaggle/input/asl-fingerspelling/train.csv\"\n\ntrain_df = pd.read_csv(TRAIN_CSV_PATH)\n\nprint(\"Total muestras:\", len(train_df))\ntrain_df.head()\n\n## para que entendamos que hay dentro de los parquets, conjunto de líneas que representan un frame identificado a la vedz por un sequence_id.\n## como veremos en cada línea del parquet hay muchas coordenadas, de cara, mano izqueirda, mano derecha, etc, a nosotros solo nos va a interesar\n#coordenadas de la mano derecha que supuestamente es la que va a representar el lenguaje de signos\n\n## el dataset se divide en parquets por temas de eficiencia y porque sería poco eficiente almacenar todo en un solo parquet, las consultas \n## a las diferentes secuencias serían muy lentas\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:15:38.56567Z","iopub.execute_input":"2026-02-24T20:15:38.566455Z","iopub.status.idle":"2026-02-24T20:15:38.778345Z","shell.execute_reply.started":"2026-02-24T20:15:38.566427Z","shell.execute_reply":"2026-02-24T20:15:38.777689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Vamos a hacer una función que filtre el \"phrase\" del dataset para que recibamos una única palabra i mediante una expresión regultar\n#que nos saque solo palabras que contengas letras del abecedario, sin números ni carácteres raros.\n\nimport re\n\ndef is_single_word(word):\n    return bool(re.fullmatch(r\"[A-Za-z]+\", word))\n\ntrain_df[\"is_single_word\"] = train_df[\"phrase\"].apply(is_single_word)\n\nsingle_word_df = train_df[train_df[\"is_single_word\"]].copy()\n\nprint(\"Total muestras:\", len(train_df))\nprint(\"Solo palabras:\", len(single_word_df))\nprint(\"Porcentaje:\", len(single_word_df) / len(train_df) * 100)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:15:40.963537Z","iopub.execute_input":"2026-02-24T20:15:40.963808Z","iopub.status.idle":"2026-02-24T20:15:41.026626Z","shell.execute_reply.started":"2026-02-24T20:15:40.963786Z","shell.execute_reply":"2026-02-24T20:15:41.026033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#No hay suficientes palabras sueltas para considerar únicament las muestras de una palabra. Como vemos, una única palabra son solo el 1% del dataset\n# y seguramente de estas palabras sueltas algunas contendran caràcteres especiales y números que queremos excluir, por lo que no usaremos este método\n\n#Como lo que queremos es detectar letra a letra, creemos que no es crítico\n\ntrain_df[\"phrase\"].head(20)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:15:44.477866Z","iopub.execute_input":"2026-02-24T20:15:44.478407Z","iopub.status.idle":"2026-02-24T20:15:44.484018Z","shell.execute_reply.started":"2026-02-24T20:15:44.478377Z","shell.execute_reply":"2026-02-24T20:15:44.483488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Como vemos, aparecen muchos carácteres especiales o números, deberíamos limpiar estas entradas.\n\nfrom collections import Counter\nimport string\n\nletters = set(string.ascii_lowercase)\n\nletter_counter = Counter()\n\nfor phrase in train_df[\"phrase\"].astype(str):\n    for ch in phrase.lower():\n        if ch in letters:\n            letter_counter[ch] += 1\n\nletter_counter","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:15:47.315354Z","iopub.execute_input":"2026-02-24T20:15:47.316013Z","iopub.status.idle":"2026-02-24T20:15:47.536753Z","shell.execute_reply.started":"2026-02-24T20:15:47.315984Z","shell.execute_reply":"2026-02-24T20:15:47.536171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(letter_counter), letter_counter.most_common()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:15:49.538233Z","iopub.execute_input":"2026-02-24T20:15:49.538519Z","iopub.status.idle":"2026-02-24T20:15:49.543888Z","shell.execute_reply.started":"2026-02-24T20:15:49.538495Z","shell.execute_reply":"2026-02-24T20:15:49.543286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Podemos ver aquí que tenemos muestras que incluyen las 26 letras del abecedario y con bastantes apariciones, por ejemplo la letra e aparece en 71986 veces\n# o la que menos, la q con 1114.\nimport re\n\ndef is_clean_phrase(phrase):\n    return bool(re.fullmatch(r\"[A-Za-z ]+\", phrase))\n\nclean_df = train_df[train_df[\"phrase\"].apply(is_clean_phrase)].copy()\n\nprint(\"Muestras limpias:\", len(clean_df))\nprint(\"Porcentaje:\", len(clean_df) / len(train_df) * 100)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:15:51.784561Z","iopub.execute_input":"2026-02-24T20:15:51.784862Z","iopub.status.idle":"2026-02-24T20:15:51.839292Z","shell.execute_reply.started":"2026-02-24T20:15:51.784837Z","shell.execute_reply":"2026-02-24T20:15:51.838675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"clean_df.iloc[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:15:55.978048Z","iopub.execute_input":"2026-02-24T20:15:55.978427Z","iopub.status.idle":"2026-02-24T20:15:55.984479Z","shell.execute_reply.started":"2026-02-24T20:15:55.978401Z","shell.execute_reply":"2026-02-24T20:15:55.98379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# division por participantes, trian, evalu i test no deberían compartir los mismos participantes\nparticipants = clean_df[\"participant_id\"].unique()\nnp.random.shuffle(participants)\n\nn = len(participants)\n\ntrain_ids = participants[:int(0.8 * n)]\nval_ids   = participants[int(0.8 * n):int(0.9 * n)]\ntest_ids  = participants[int(0.9 * n):]\n\ntrain_df_split = clean_df[clean_df[\"participant_id\"].isin(train_ids)]\nval_df_split   = clean_df[clean_df[\"participant_id\"].isin(val_ids)]\ntest_df_split  = clean_df[clean_df[\"participant_id\"].isin(test_ids)]\n\nprint(\"Train:\", len(train_df_split))\nprint(\"Val:  \", len(val_df_split))\nprint(\"Test: \", len(test_df_split))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:15:58.790576Z","iopub.execute_input":"2026-02-24T20:15:58.791265Z","iopub.status.idle":"2026-02-24T20:15:58.80692Z","shell.execute_reply.started":"2026-02-24T20:15:58.791237Z","shell.execute_reply":"2026-02-24T20:15:58.806326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"VOCAB_PATH = \"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\"\n\nwith open(VOCAB_PATH) as f:\n    original_letter_to_int = json.load(f)\n\nprint(\"Tamaño vocabulario original:\", len(original_letter_to_int))\nprint(list(original_letter_to_int.items()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:00.787839Z","iopub.execute_input":"2026-02-24T20:16:00.788602Z","iopub.status.idle":"2026-02-24T20:16:00.800312Z","shell.execute_reply.started":"2026-02-24T20:16:00.788571Z","shell.execute_reply":"2026-02-24T20:16:00.799688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Añadimos el blank para que CTC pueda detectar cuando hay un cambio de letra\nletter_to_int = {}\n\nletter_to_int[\"<blank>\"] = 0\n\n# Desplazamos el resto +1\nfor char, idx in original_letter_to_int.items():\n    letter_to_int[char] = idx + 1\n\nint_to_letter = {v: k for k, v in letter_to_int.items()}\n\nprint(\"Tamaño vocabulario CTC:\", len(letter_to_int))\nprint(list(letter_to_int.items()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:02.983099Z","iopub.execute_input":"2026-02-24T20:16:02.983429Z","iopub.status.idle":"2026-02-24T20:16:02.988668Z","shell.execute_reply.started":"2026-02-24T20:16:02.983405Z","shell.execute_reply":"2026-02-24T20:16:02.987955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Nuevo vocabulario filtrado por las letras que vamos a procesar al principio, que no contendran números ni carácteres especiales, por lo que este\n#se reduce a 27 i el Blank, que es lo que el ctc detecta como frames donde no pasa nada, pausas entre letras, transiciones, ruido...\n\nimport string\n\n# letras permitidas\nletters = list(string.ascii_lowercase)\n\n# vocabulario CTC\nletter_to_int = {\"<blank>\": 0}\n\nfor i, ch in enumerate(letters):\n    letter_to_int[ch] = i + 1\n\nint_to_letter = {v: k for k, v in letter_to_int.items()}\n\nprint(\"Tamaño vocabulario:\", len(letter_to_int))\nprint(letter_to_int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:05.61268Z","iopub.execute_input":"2026-02-24T20:16:05.613306Z","iopub.status.idle":"2026-02-24T20:16:05.618471Z","shell.execute_reply.started":"2026-02-24T20:16:05.613277Z","shell.execute_reply":"2026-02-24T20:16:05.617765Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Encode target","metadata":{}},{"cell_type":"code","source":"def encode_phrase(phrase, letter_to_int):\n    return [letter_to_int[c] for c in phrase if c in letter_to_int]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:08.393034Z","iopub.execute_input":"2026-02-24T20:16:08.393357Z","iopub.status.idle":"2026-02-24T20:16:08.397192Z","shell.execute_reply.started":"2026-02-24T20:16:08.393331Z","shell.execute_reply":"2026-02-24T20:16:08.39652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Añadir frases codificadas\ntrain_df_split = train_df_split.copy()\nval_df_split   = val_df_split.copy()\n\ntrain_df_split[\"encoded\"] = train_df_split[\"phrase\"].apply(\n    lambda x: encode_phrase(x, letter_to_int)\n)\n\nval_df_split[\"encoded\"] = val_df_split[\"phrase\"].apply(\n    lambda x: encode_phrase(x, letter_to_int)\n)\n\ntrain_df_split[[\"phrase\", \"encoded\"]].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:09.376395Z","iopub.execute_input":"2026-02-24T20:16:09.376688Z","iopub.status.idle":"2026-02-24T20:16:09.405008Z","shell.execute_reply.started":"2026-02-24T20:16:09.376661Z","shell.execute_reply":"2026-02-24T20:16:09.404269Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Normalize inputs","metadata":{}},{"cell_type":"code","source":"MAX_FRAMES = 160\nRIGHT_HAND_COLS = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:13.672724Z","iopub.execute_input":"2026-02-24T20:16:13.673284Z","iopub.status.idle":"2026-02-24T20:16:13.676691Z","shell.execute_reply.started":"2026-02-24T20:16:13.673254Z","shell.execute_reply":"2026-02-24T20:16:13.676012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def read_right_hand_sequence(file_id, sequence_id):\n#     global RIGHT_HAND_COLS\n\n#     path = f\"/kaggle/input/asl-fingerspelling/train_landmarks/{file_id}.parquet\"\n#     pq_file = pq.ParquetFile(path)\n\n#     # Detectamos columnas una sola vez\n#     if RIGHT_HAND_COLS is None:\n#         RIGHT_HAND_COLS = [c for c in pq_file.schema.names if \"right_hand\" in c]\n\n#     table = pq.read_table(\n#         path,\n#         filters=[(\"sequence_id\", \"=\", sequence_id)],\n#         columns=RIGHT_HAND_COLS\n#     )\n    \n#     X = table.to_pandas().values.astype(np.float32)\n#     return X\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T07:46:50.685044Z","iopub.execute_input":"2026-02-23T07:46:50.685534Z","iopub.status.idle":"2026-02-23T07:46:50.690258Z","shell.execute_reply.started":"2026-02-23T07:46:50.685506Z","shell.execute_reply":"2026-02-23T07:46:50.689632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_right_hand_sequence_df(file_id, sequence_id):\n    global RIGHT_HAND_COLS\n\n    path = f\"/kaggle/input/asl-fingerspelling/train_landmarks/{file_id}.parquet\"\n    pq_file = pq.ParquetFile(path)\n\n    # Detectamos columnas una sola vez\n    if RIGHT_HAND_COLS is None:\n        RIGHT_HAND_COLS = [c for c in pq_file.schema.names if \"right_hand\" in c]\n\n    table = pq.read_table(\n        path,\n        filters=[(\"sequence_id\", \"=\", sequence_id)],\n        columns=RIGHT_HAND_COLS\n    )\n\n    X = table.to_pandas()\n    return X","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:19.247047Z","iopub.execute_input":"2026-02-24T20:16:19.2478Z","iopub.status.idle":"2026-02-24T20:16:19.252837Z","shell.execute_reply.started":"2026-02-24T20:16:19.247771Z","shell.execute_reply":"2026-02-24T20:16:19.252039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_right_hand_sequence_array(file_id, sequence_id):\n    return read_right_hand_sequence_df(file_id, sequence_id).values.astype(np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:20.865369Z","iopub.execute_input":"2026-02-24T20:16:20.865653Z","iopub.status.idle":"2026-02-24T20:16:20.869312Z","shell.execute_reply.started":"2026-02-24T20:16:20.865627Z","shell.execute_reply":"2026-02-24T20:16:20.86873Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Center hand","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:23.011591Z","iopub.execute_input":"2026-02-24T20:16:23.011889Z","iopub.status.idle":"2026-02-24T20:16:25.500398Z","shell.execute_reply.started":"2026-02-24T20:16:23.011862Z","shell.execute_reply":"2026-02-24T20:16:25.499834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def center_wrist(pdf):\n    tmp = pdf.copy()\n    for i in range (0,21):\n        tmp[f\"x_right_hand_{i}\"] -= pdf[\"x_right_hand_0\"]\n        tmp[f\"y_right_hand_{i}\"] -= pdf[\"y_right_hand_0\"]\n        tmp[f\"z_right_hand_{i}\"] -= pdf[\"z_right_hand_0\"]\n    return tmp","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:28.269035Z","iopub.execute_input":"2026-02-24T20:16:28.269753Z","iopub.status.idle":"2026-02-24T20:16:28.273783Z","shell.execute_reply.started":"2026-02-24T20:16:28.269721Z","shell.execute_reply":"2026-02-24T20:16:28.273194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def normalize_to_box(pdf):\n    x_min = pdf[[f\"x_right_hand_{i}\" for i in range (0,21)]].min().min()\n    x_max = pdf[[f\"x_right_hand_{i}\" for i in range (0,21)]].max().max()\n    y_min = pdf[[f\"y_right_hand_{i}\" for i in range (0,21)]].min().min()\n    y_max = pdf[[f\"y_right_hand_{i}\" for i in range (0,21)]].max().max()\n    \n    ratio = min(1/(x_max-x_min), 1/(y_max-y_min))\n    tmp = pdf.copy()\n    for i in range (0,21):\n        tmp[f\"x_right_hand_{i}\"] -= x_min\n        tmp[f\"x_right_hand_{i}\"] *= ratio\n        tmp[f\"y_right_hand_{i}\"] -= y_min\n        tmp[f\"y_right_hand_{i}\"] *= ratio\n        tmp[f\"z_right_hand_{i}\"] *= ratio\n    return tmp","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:30.101641Z","iopub.execute_input":"2026-02-24T20:16:30.101927Z","iopub.status.idle":"2026-02-24T20:16:30.107882Z","shell.execute_reply.started":"2026-02-24T20:16:30.101901Z","shell.execute_reply":"2026-02-24T20:16:30.107201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"row = train_df_split.iloc[20]\nX_pdf = read_right_hand_sequence_df(row[\"file_id\"], row[\"sequence_id\"])\nX_pdf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:31.557271Z","iopub.execute_input":"2026-02-24T20:16:31.557852Z","iopub.status.idle":"2026-02-24T20:16:32.086962Z","shell.execute_reply.started":"2026-02-24T20:16:31.557825Z","shell.execute_reply":"2026-02-24T20:16:32.086317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# distance between hand landmark 0 and 9 should be normalized, doesn't seem to be the case here, ToDo ?\n#(cf. https://storage.googleapis.com/mediapipe-assets/Model%20Card%20Hand%20Tracking%20(Lite_Full)%20with%20Fairness%20Oct%202021.pdf)\n\ndef dist(df, i,j):\n    return np.linalg.norm(df[[f\"{ax}_right_hand_{i}\" for ax in ['x', 'y', 'z']]].values - df[[f\"{ax}_right_hand_{j}\" for ax in ['x', 'y', 'z']]].values, axis=1)\n\n#X_pd[~X_pd.x_right_hand_0.isna()][['x_right_hand_0', 'y_right_hand_0', 'z_right_hand_0']]\nsns.displot(dist(X_pdf[~X_pdf['x_right_hand_0'].isna()], 0, 9))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:34.487422Z","iopub.execute_input":"2026-02-24T20:16:34.488028Z","iopub.status.idle":"2026-02-24T20:16:34.766189Z","shell.execute_reply.started":"2026-02-24T20:16:34.487999Z","shell.execute_reply":"2026-02-24T20:16:34.765611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_centered = center_wrist(X_pdf)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:37.584531Z","iopub.execute_input":"2026-02-24T20:16:37.585087Z","iopub.status.idle":"2026-02-24T20:16:37.603957Z","shell.execute_reply.started":"2026-02-24T20:16:37.585055Z","shell.execute_reply":"2026-02-24T20:16:37.603215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.displot(X_pdf[[f\"x_right_hand_{i}\" for i in [4, 8, 12, 16, 20]]], kind='kde')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:38.453327Z","iopub.execute_input":"2026-02-24T20:16:38.453622Z","iopub.status.idle":"2026-02-24T20:16:38.822912Z","shell.execute_reply.started":"2026-02-24T20:16:38.453596Z","shell.execute_reply":"2026-02-24T20:16:38.822212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.displot(X_centered[[f\"x_right_hand_{i}\" for i in [4, 8, 12, 16, 20]]], kind='kde')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:43.50444Z","iopub.execute_input":"2026-02-24T20:16:43.50475Z","iopub.status.idle":"2026-02-24T20:16:43.84698Z","shell.execute_reply.started":"2026-02-24T20:16:43.504722Z","shell.execute_reply":"2026-02-24T20:16:43.846384Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Visualization","metadata":{}},{"cell_type":"code","source":"!pip install mediapipe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:48.654637Z","iopub.execute_input":"2026-02-24T20:16:48.655335Z","iopub.status.idle":"2026-02-24T20:16:53.171043Z","shell.execute_reply.started":"2026-02-24T20:16:48.655304Z","shell.execute_reply":"2026-02-24T20:16:53.16997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_hands(seq_df):\n    import mediapipe as mp\n    from mediapipe.tasks.python.components.containers.landmark import NormalizedLandmark, Landmark\n    #import numpy as np\n    \n    mp_hands = mp.tasks.vision.HandLandmarksConnections\n    mp_drawing = mp.tasks.vision.drawing_utils\n    mp_drawing_styles = mp.tasks.vision.drawing_styles\n    \n    images = []\n    all_hand_landmarks = []\n    for seq_idx in range(len(seq_df)):\n        x_pose = seq_df.iloc[seq_idx].filter(regex=\"x_right_hand.*\").values\n        y_pose = seq_df.iloc[seq_idx].filter(regex=\"y_right_hand.*\").values\n        z_pose = seq_df.iloc[seq_idx].filter(regex=\"z_right_hand.*\").values\n\n        right_hand_landmarks = [NormalizedLandmark(x, y, z) for x, y, z in zip(x_pose, y_pose, z_pose)]\n\n        right_hand_image = np.zeros((900, 600, 3), dtype=\"int8\")\n\n        mp_drawing.draw_landmarks(\n              right_hand_image,\n              right_hand_landmarks,\n              mp_hands.HAND_CONNECTIONS,\n              mp_drawing_styles.get_default_hand_landmarks_style(),\n              mp_drawing_styles.get_default_hand_connections_style())\n        \n        x_pose = seq_df.iloc[seq_idx].filter(regex=\"x_left_hand.*\").values\n        y_pose = seq_df.iloc[seq_idx].filter(regex=\"y_left_hand.*\").values\n        z_pose = seq_df.iloc[seq_idx].filter(regex=\"z_left_hand.*\").values\n\n        left_hand_landmarks = [NormalizedLandmark(x, y, z) for x, y, z in zip(x_pose, y_pose, z_pose)]\n        \n        left_hand_image = np.zeros((900, 600, 3), dtype=\"int8\")\n\n        mp_drawing.draw_landmarks(\n                left_hand_image,\n                left_hand_landmarks,\n                mp_hands.HAND_CONNECTIONS,\n                mp_drawing_styles.get_default_hand_landmarks_style(),\n                mp_drawing_styles.get_default_hand_connections_style())\n        \n        images.append([right_hand_image.astype(np.uint8), left_hand_image.astype(np.uint8)])\n        all_hand_landmarks.append([right_hand_landmarks, left_hand_landmarks])\n    return images, all_hand_landmarks","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:53.172894Z","iopub.execute_input":"2026-02-24T20:16:53.173174Z","iopub.status.idle":"2026-02-24T20:16:53.182574Z","shell.execute_reply.started":"2026-02-24T20:16:53.173116Z","shell.execute_reply":"2026-02-24T20:16:53.181996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from matplotlib import animation, rc\nrc('animation', html='jshtml')\n\ndef create_animation(images):\n    fig = plt.figure(figsize=(2, 3))\n    ax = plt.Axes(fig, [0., 0., 1., 1.])\n    ax.set_axis_off()\n    fig.add_axes(ax)\n    im=ax.imshow(images[0], cmap=\"gray\")\n    plt.close(fig)\n    \n    def animate_func(i):\n        im.set_array(images[i])\n        return [im]\n\n    return animation.FuncAnimation(fig, animate_func, frames=len(images), interval=1000/10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:56.117379Z","iopub.execute_input":"2026-02-24T20:16:56.117959Z","iopub.status.idle":"2026-02-24T20:16:56.137923Z","shell.execute_reply.started":"2026-02-24T20:16:56.117929Z","shell.execute_reply":"2026-02-24T20:16:56.137404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hand_images, hand_landmarks = get_hands(X_pdf)\ncreate_animation(np.array(hand_images)[:, 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:16:58.270011Z","iopub.execute_input":"2026-02-24T20:16:58.270741Z","iopub.status.idle":"2026-02-24T20:17:23.12093Z","shell.execute_reply.started":"2026-02-24T20:16:58.270712Z","shell.execute_reply":"2026-02-24T20:17:23.120018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hand_images, hand_landmarks = get_hands(normalize_to_box(X_pdf))\ncreate_animation(np.array(hand_images)[:, 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:17:39.049567Z","iopub.execute_input":"2026-02-24T20:17:39.050838Z","iopub.status.idle":"2026-02-24T20:17:44.33963Z","shell.execute_reply.started":"2026-02-24T20:17:39.050789Z","shell.execute_reply":"2026-02-24T20:17:44.338839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hand_images, hand_landmarks = get_hands(normalize_to_box(X_centered))\ncreate_animation(np.array(hand_images)[:, 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:17:58.944055Z","iopub.execute_input":"2026-02-24T20:17:58.944845Z","iopub.status.idle":"2026-02-24T20:18:04.346498Z","shell.execute_reply.started":"2026-02-24T20:17:58.944814Z","shell.execute_reply":"2026-02-24T20:18:04.345633Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Normalize sequence length","metadata":{}},{"cell_type":"code","source":"Aqui de vuelta todo lo anterior","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def normalize_frames(X, max_frames=MAX_FRAMES):\n    T, D = X.shape\n\n    if T > max_frames:\n        return X[:max_frames]\n\n    if T < max_frames:\n        pad = np.zeros((max_frames - T, D), dtype=np.float32)\n        return np.vstack([X, pad])\n\n    return X","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:18:27.804305Z","iopub.execute_input":"2026-02-24T20:18:27.80497Z","iopub.status.idle":"2026-02-24T20:18:27.809058Z","shell.execute_reply.started":"2026-02-24T20:18:27.804928Z","shell.execute_reply":"2026-02-24T20:18:27.808393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#row = train_df_split.iloc[20]\n\n#X = read_right_hand_sequence(row[\"file_id\"], row[\"sequence_id\"])\nX = X_centered.values.astype(np.float32)\nXn = normalize_frames(X)\n\nprint(\"Original:\", X.shape)\nprint(\"Normalizado:\", Xn.shape)\nprint(Xn[:2])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:19:21.687601Z","iopub.execute_input":"2026-02-24T20:19:21.68816Z","iopub.status.idle":"2026-02-24T20:19:21.694304Z","shell.execute_reply.started":"2026-02-24T20:19:21.688112Z","shell.execute_reply":"2026-02-24T20:19:21.693595Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"row = train_df_split.iloc[0]\nfile_id, sequence_id = row[\"file_id\"], row[\"sequence_id\"]\n\npath = f\"/kaggle/input/asl-fingerspelling/train_landmarks/{file_id}.parquet\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:19:59.33995Z","iopub.execute_input":"2026-02-24T20:19:59.340519Z","iopub.status.idle":"2026-02-24T20:19:59.344995Z","shell.execute_reply.started":"2026-02-24T20:19:59.340487Z","shell.execute_reply":"2026-02-24T20:19:59.344363Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pq_file = pq.ParquetFile(path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:20:10.595627Z","iopub.execute_input":"2026-02-24T20:20:10.595916Z","iopub.status.idle":"2026-02-24T20:20:10.621708Z","shell.execute_reply.started":"2026-02-24T20:20:10.595893Z","shell.execute_reply":"2026-02-24T20:20:10.621103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def count_valid_frames(X):\n    # Un frame es válido si NO todo es NaN\n    return np.sum(~np.all(np.isnan(X), axis=1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:20:28.087881Z","iopub.execute_input":"2026-02-24T20:20:28.088199Z","iopub.status.idle":"2026-02-24T20:20:28.09209Z","shell.execute_reply.started":"2026-02-24T20:20:28.088172Z","shell.execute_reply":"2026-02-24T20:20:28.091533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nclass ASLRightHandDataset(Dataset):\n    def __init__(self, df, max_frames=160):\n        self.df = df.reset_index(drop=True)\n        self.max_frames = max_frames\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n\n        X_raw = read_right_hand_sequence_df(\n            row[\"file_id\"],\n            row[\"sequence_id\"]\n        )  \n        X_centered = center_wrist(X_raw)\n    \n        input_len = count_valid_frames(X_centered)\n    \n        Y = torch.tensor(row[\"encoded\"], dtype=torch.long)\n        target_len = len(Y)\n\n        if input_len < target_len:\n            return None\n\n\n        X = normalize_frames(X_centered, self.max_frames)\n    \n        X = np.nan_to_num(X, nan=0.0)\n    \n        X = torch.tensor(X, dtype=torch.float32)\n    \n        input_len = min(input_len, self.max_frames)\n    \n        return X, Y, input_len, target_len\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:21:59.181231Z","iopub.execute_input":"2026-02-24T20:21:59.181537Z","iopub.status.idle":"2026-02-24T20:21:59.187971Z","shell.execute_reply.started":"2026-02-24T20:21:59.181512Z","shell.execute_reply":"2026-02-24T20:21:59.187218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def collate_fn(batch):\n    # Quitamos ejemplos inválidos (None)\n    batch = [b for b in batch if b is not None]\n\n    # Si todo el batch era inválido, devolvemos None\n    if len(batch) == 0:\n        return None\n\n    Xs = []\n    Ys = []\n    in_lens = []\n    tar_lens = []\n\n    for X, Y, in_len, tar_len in batch:\n        Xs.append(X)\n        Ys.append(Y)\n        in_lens.append(in_len)\n        tar_lens.append(tar_len)\n\n    Xs = torch.stack(Xs)\n    Ys = torch.cat(Ys)\n    in_lens = torch.tensor(in_lens, dtype=torch.long)\n    tar_lens = torch.tensor(tar_lens, dtype=torch.long)\n\n    return Xs, Ys, in_lens, tar_lens\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:22:01.739112Z","iopub.execute_input":"2026-02-24T20:22:01.739702Z","iopub.status.idle":"2026-02-24T20:22:01.744649Z","shell.execute_reply.started":"2026-02-24T20:22:01.739672Z","shell.execute_reply":"2026-02-24T20:22:01.743964Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df_tiny = train_df_split.sample(200, random_state=0)\ntrain_dataset_tiny = ASLRightHandDataset(train_df_tiny)\n\nprint(\"INFO: Creating data loader with TINY size (200)\")\n\ntrain_loader_tiny = DataLoader(\n    train_dataset_tiny,\n    batch_size=4,\n    shuffle=True,\n    collate_fn=collate_fn,\n    num_workers=0\n)\n\nprint(\"INFO: Checking sizes\")\n\nX, Y, in_len, tar_len = next(iter(train_loader_tiny))\n\nprint(\"X:\", X.shape)          # (B, 160, 63)\nprint(\"Y:\", Y.shape)          # (sum of target lengths)\nprint(\"in_len:\", in_len)\nprint(\"tar_len:\", tar_len)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:22:43.014395Z","iopub.execute_input":"2026-02-24T20:22:43.014693Z","iopub.status.idle":"2026-02-24T20:22:44.033722Z","shell.execute_reply.started":"2026-02-24T20:22:43.014666Z","shell.execute_reply":"2026-02-24T20:22:44.032985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class EmbeddedRNN(nn.Module):\n    def __init__(self, input_size, embed_size, hidden_size, num_classes):\n        super().__init__()\n\n        self.embedding = nn.Linear(input_size, embed_size)\n\n        self.rnn = nn.RNN(\n            input_size=embed_size,\n            hidden_size=hidden_size,\n            batch_first=True\n        )\n\n        self.classifier = nn.Linear(hidden_size, num_classes)\n        self.log_softmax = nn.LogSoftmax(dim=-1)\n\n    def forward(self, x):\n\n        x = self.embedding(x)       \n        out, _ = self.rnn(x)      \n        out = self.classifier(out)  \n        out = self.log_softmax(out)\n        out = out.permute(1, 0, 2)  \n\n        return out\nprint(\"INFO: EmbeddedRNN defined\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:23:13.489296Z","iopub.execute_input":"2026-02-24T20:23:13.489926Z","iopub.status.idle":"2026-02-24T20:23:13.495792Z","shell.execute_reply.started":"2026-02-24T20:23:13.489894Z","shell.execute_reply":"2026-02-24T20:23:13.494992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_classes = len(letter_to_int)\n\nmodel = EmbeddedRNN(\n    input_size=63,         \n    embed_size=128,         \n    hidden_size=128,\n    num_classes=len(letter_to_int)\n).to(DEVICE)\n\nprint(f\"Model: EmbeddedRNN / Parameters: {sum(p.numel() for p in model.parameters()):,}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:23:16.208634Z","iopub.execute_input":"2026-02-24T20:23:16.209399Z","iopub.status.idle":"2026-02-24T20:23:16.545787Z","shell.execute_reply.started":"2026-02-24T20:23:16.209356Z","shell.execute_reply":"2026-02-24T20:23:16.545165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.eval()\n\nwith torch.no_grad():\n    outputs = model(X.to(DEVICE))\n\nprint(outputs.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:23:27.429546Z","iopub.execute_input":"2026-02-24T20:23:27.42983Z","iopub.status.idle":"2026-02-24T20:23:27.84568Z","shell.execute_reply.started":"2026-02-24T20:23:27.429804Z","shell.execute_reply":"2026-02-24T20:23:27.845033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"criterion = nn.CTCLoss(\n    blank=letter_to_int[\"<blank>\"],\n    zero_infinity=True,\n    reduction='mean'  # ensure proper reduction\n)\n\nloss = criterion(\n    outputs,    # (T, B, C)\n    Y.to(DEVICE),\n    in_len,\n    tar_len\n)\n\nprint(\"CTC loss:\", loss.item())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:23:37.01664Z","iopub.execute_input":"2026-02-24T20:23:37.01691Z","iopub.status.idle":"2026-02-24T20:23:37.243788Z","shell.execute_reply.started":"2026-02-24T20:23:37.016888Z","shell.execute_reply":"2026-02-24T20:23:37.24299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_one_epoch(model, loader, optimizer, criterion, device):\n    model.train()\n    total_loss = 0.0\n    num_batches = 0\n\n    for batch in loader:\n\n        if batch is None:\n            continue\n\n        X, Y, input_lens, target_lens = batch\n\n        X = X.to(device)\n        Y = Y.to(device)\n        input_lens = input_lens.to(device)\n        target_lens = target_lens.to(device)\n\n        optimizer.zero_grad()\n\n        outputs = model(X)  # (T, B, C)\n\n        loss = criterion(\n            outputs,\n            Y,\n            input_lens,\n            target_lens\n        )\n\n        loss.backward()\n\n\n        optimizer.step()\n\n        total_loss += loss.item()\n        num_batches += 1\n\n    avg_loss = total_loss / num_batches\n\n    return avg_loss\nprint(\"INFO: train_one_epoch method defined\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:23:44.356535Z","iopub.execute_input":"2026-02-24T20:23:44.356839Z","iopub.status.idle":"2026-02-24T20:23:44.363011Z","shell.execute_reply.started":"2026-02-24T20:23:44.35681Z","shell.execute_reply":"2026-02-24T20:23:44.362323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import time\nstart_time = time.time()\n\nnum_epochs = 5\ntrain_losses = []\ngrad_norms = []\nlrate = 1e-3\noptimizer = optim.Adam(model.parameters(), lr=lrate)\n\nprint(\"=\"*60)\nprint(f\"INITIAL TRAINING\")\nprint(\"=\"*60)\n\n# Start a new wandb run to track this training script\nrun = wandb.init(\n    # Team workspace\n    entity=\"inaki-rodriguez-reyes-upc-universidad-peruana-de-ciencia\",\n    # Preview project\n    project=\"asl-fingerspelling-previews\",\n    \n    # Track hyperparameters and run metadata.\n    config={\n        \"name\": \"INITIAL TRAINING (CENTER+WORDS+WB)\",\n        \"learning_rate\": lrate,\n        \"architecture\": \"EmbeddedRNN\",\n        \"dataset\": \"Google-ASL\",\n        \"epochs\": num_epochs,\n    },\n)\n\n\nprint(f\"[Very first training with TINY / lr={lrate}]\")\nfor epoch in range(num_epochs):\n    int_time = time.time()\n    loss = train_one_epoch(\n        model,\n        train_loader_tiny,\n        optimizer,\n        criterion,\n        DEVICE\n    )\n    train_losses.append(loss)\n    print(f\"Epoch {epoch+1} -> LOSS = {loss:.2f}\")\n    run.log({\"loss\": loss})\n    end_time = time.time()\n    print(f\"Elapsed time: {end_time - int_time:.2f} secs\")\n\nrun.finish()\n\nfinal_time = time.time()\nprint(\"-\"*60)\nprint(f\"TOTAL time: {(final_time - start_time)/60:.2f} mins\")\nprint(\"-\"*60)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:24:37.208777Z","iopub.execute_input":"2026-02-24T20:24:37.209397Z","iopub.status.idle":"2026-02-24T20:26:48.288838Z","shell.execute_reply.started":"2026-02-24T20:24:37.209368Z","shell.execute_reply":"2026-02-24T20:26:48.288072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(6,4))\nplt.plot(train_losses, marker=\"o\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"CTC Loss\")\nplt.title(f\"Training Loss over {num_epochs} Epochs / lr={lrate}\")\nplt.grid(True)\nplt.show()\n\nprint(\"-\"*60)\nprint(f\"LOSS TREND\")\nprint(f\"First loss: {train_losses[0]:.2f}\")\nprint(f\"Last loss: {train_losses[-1]:.2f}\")\nprint(f\"Loss decreased: {train_losses[0] - train_losses[-1]:.2f}\")\nprint(\"-\"*60)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T20:27:00.479202Z","iopub.execute_input":"2026-02-24T20:27:00.479511Z","iopub.status.idle":"2026-02-24T20:27:00.609221Z","shell.execute_reply.started":"2026-02-24T20:27:00.479485Z","shell.execute_reply":"2026-02-24T20:27:00.608592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"INFO: [END_OF_NOTEBOOK]\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-22T19:44:32.367815Z","iopub.execute_input":"2026-02-22T19:44:32.368068Z","iopub.status.idle":"2026-02-22T19:44:32.371565Z","shell.execute_reply.started":"2026-02-22T19:44:32.368044Z","shell.execute_reply":"2026-02-22T19:44:32.370819Z"}},"outputs":[],"execution_count":null}]}