{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":52950,"databundleVersionId":5973250}],"dockerImageVersionId":31259,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2026-02-23T23:23:44.187238Z","iopub.execute_input":"2026-02-23T23:23:44.187717Z","iopub.status.idle":"2026-02-23T23:23:44.194625Z","shell.execute_reply.started":"2026-02-23T23:23:44.187672Z","shell.execute_reply":"2026-02-23T23:23:44.193338Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport json\nimport numpy as np\nimport pandas as pd\n\n# Lectura de parquets \nimport pyarrow.parquet as pq\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\n\nSEED = 42\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\n\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:23:44.196469Z","iopub.execute_input":"2026-02-23T23:23:44.196857Z","iopub.status.idle":"2026-02-23T23:23:46.748268Z","shell.execute_reply.started":"2026-02-23T23:23:44.196826Z","shell.execute_reply":"2026-02-23T23:23:46.747301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_CSV_PATH = \"/kaggle/input/asl-fingerspelling/train.csv\"\n\ntrain_df = pd.read_csv(TRAIN_CSV_PATH)\n\nprint(\"Total muestras:\", len(train_df))\ntrain_df.head()\n\n## para que entendamos que hay dentro de los parquets, conjunto de líneas que representan un frame identificado a la vedz por un sequence_id.\n## como veremos en cada línea del parquet hay muchas coordenadas, de cara, mano izqueirda, mano derecha, etc, a nosotros solo nos va a interesar\n#coordenadas de la mano derecha que supuestamente es la que va a representar el lenguaje de signos\n\n## el dataset se divide en parquets por temas de eficiencia y porque sería poco eficiente almacenar todo en un solo parquet, las consultas \n## a las diferentes secuencias serían muy lentas\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:23:46.75026Z","iopub.execute_input":"2026-02-23T23:23:46.750882Z","iopub.status.idle":"2026-02-23T23:23:46.903242Z","shell.execute_reply.started":"2026-02-23T23:23:46.750848Z","shell.execute_reply":"2026-02-23T23:23:46.902439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#división por participantes, trian, evalu i test no deberían compartir los mismos participantes\nparticipants = train_df[\"participant_id\"].unique()\nnp.random.shuffle(participants)\n\nn = len(participants)\n\ntrain_ids = participants[:int(0.8 * n)]\nval_ids   = participants[int(0.8 * n):int(0.9 * n)]\ntest_ids  = participants[int(0.9 * n):]\n\ntrain_df_split = train_df[train_df[\"participant_id\"].isin(train_ids)]\nval_df_split   = train_df[train_df[\"participant_id\"].isin(val_ids)]\ntest_df_split  = train_df[train_df[\"participant_id\"].isin(test_ids)]\n\nprint(\"Train:\", len(train_df_split))\nprint(\"Val:  \", len(val_df_split))\nprint(\"Test: \", len(test_df_split))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:23:46.904238Z","iopub.execute_input":"2026-02-23T23:23:46.904639Z","iopub.status.idle":"2026-02-23T23:23:46.932104Z","shell.execute_reply.started":"2026-02-23T23:23:46.904599Z","shell.execute_reply":"2026-02-23T23:23:46.931123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"VOCAB_PATH = \"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\"\n\nwith open(VOCAB_PATH) as f:\n    original_letter_to_int = json.load(f)\n\nprint(\"Tamaño vocabulario original:\", len(original_letter_to_int))\nprint(list(original_letter_to_int.items()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:23:46.934689Z","iopub.execute_input":"2026-02-23T23:23:46.935023Z","iopub.status.idle":"2026-02-23T23:23:46.943006Z","shell.execute_reply.started":"2026-02-23T23:23:46.934994Z","shell.execute_reply":"2026-02-23T23:23:46.941955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Añadimos el blank para que CTC pueda detectar cuando hay un cambio de letra\nletter_to_int = {}\n\nletter_to_int[\"<blank>\"] = 0\n\n# Desplazamos el resto +1\nfor char, idx in original_letter_to_int.items():\n    letter_to_int[char] = idx + 1\n\nint_to_letter = {v: k for k, v in letter_to_int.items()}\n\nprint(\"Tamaño vocabulario CTC:\", len(letter_to_int))\nprint(list(letter_to_int.items()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:23:46.944425Z","iopub.execute_input":"2026-02-23T23:23:46.944761Z","iopub.status.idle":"2026-02-23T23:23:46.960503Z","shell.execute_reply.started":"2026-02-23T23:23:46.944731Z","shell.execute_reply":"2026-02-23T23:23:46.959504Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Encode target","metadata":{}},{"cell_type":"code","source":"def encode_phrase(phrase, letter_to_int):\n    return [letter_to_int[c] for c in phrase if c in letter_to_int]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:23:46.961871Z","iopub.execute_input":"2026-02-23T23:23:46.962267Z","iopub.status.idle":"2026-02-23T23:23:46.987912Z","shell.execute_reply.started":"2026-02-23T23:23:46.962227Z","shell.execute_reply":"2026-02-23T23:23:46.986564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Añadir frases codificadas\ntrain_df_split = train_df_split.copy()\nval_df_split   = val_df_split.copy()\n\ntrain_df_split[\"encoded\"] = train_df_split[\"phrase\"].apply(\n    lambda x: encode_phrase(x, letter_to_int)\n)\n\nval_df_split[\"encoded\"] = val_df_split[\"phrase\"].apply(\n    lambda x: encode_phrase(x, letter_to_int)\n)\n\ntrain_df_split[[\"phrase\", \"encoded\"]].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:23:46.98885Z","iopub.execute_input":"2026-02-23T23:23:46.989128Z","iopub.status.idle":"2026-02-23T23:23:47.250964Z","shell.execute_reply.started":"2026-02-23T23:23:46.989104Z","shell.execute_reply":"2026-02-23T23:23:47.249788Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Normalize inputs","metadata":{}},{"cell_type":"code","source":"MAX_FRAMES = 160\nRIGHT_HAND_COLS = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:24:56.80531Z","iopub.execute_input":"2026-02-23T23:24:56.805906Z","iopub.status.idle":"2026-02-23T23:24:56.812063Z","shell.execute_reply.started":"2026-02-23T23:24:56.805858Z","shell.execute_reply":"2026-02-23T23:24:56.810906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_right_hand_sequence_df(file_id, sequence_id):\n    global RIGHT_HAND_COLS\n\n    path = f\"/kaggle/input/asl-fingerspelling/train_landmarks/{file_id}.parquet\"\n    pq_file = pq.ParquetFile(path)\n\n    # Detectamos columnas una sola vez\n    if RIGHT_HAND_COLS is None:\n        RIGHT_HAND_COLS = [c for c in pq_file.schema.names if \"right_hand\" in c]\n\n    table = pq.read_table(\n        path,\n        filters=[(\"sequence_id\", \"=\", sequence_id)],\n        columns=RIGHT_HAND_COLS\n    )\n\n    X = table.to_pandas()\n    return X","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:24:56.813676Z","iopub.execute_input":"2026-02-23T23:24:56.814733Z","iopub.status.idle":"2026-02-23T23:24:56.835065Z","shell.execute_reply.started":"2026-02-23T23:24:56.81469Z","shell.execute_reply":"2026-02-23T23:24:56.833875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_right_hand_sequence_array(file_id, sequence_id):\n    return read_right_hand_sequence_df(file_id, sequence_id).values.astype(np.float32)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:24:56.836557Z","iopub.execute_input":"2026-02-23T23:24:56.836949Z","iopub.status.idle":"2026-02-23T23:24:56.858294Z","shell.execute_reply.started":"2026-02-23T23:24:56.836907Z","shell.execute_reply":"2026-02-23T23:24:56.857071Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Center hand","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:24:56.861006Z","iopub.execute_input":"2026-02-23T23:24:56.861445Z","iopub.status.idle":"2026-02-23T23:24:56.878652Z","shell.execute_reply.started":"2026-02-23T23:24:56.861388Z","shell.execute_reply":"2026-02-23T23:24:56.877411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def center_wrist(pdf):\n    tmp = pdf.copy()\n    for i in range (0,21):\n        tmp[f\"x_right_hand_{i}\"] -= pdf[\"x_right_hand_0\"]\n        tmp[f\"y_right_hand_{i}\"] -= pdf[\"y_right_hand_0\"]\n        tmp[f\"z_right_hand_{i}\"] -= pdf[\"z_right_hand_0\"]\n    return tmp","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:24:56.88017Z","iopub.execute_input":"2026-02-23T23:24:56.880639Z","iopub.status.idle":"2026-02-23T23:24:56.898037Z","shell.execute_reply.started":"2026-02-23T23:24:56.880598Z","shell.execute_reply":"2026-02-23T23:24:56.896686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def normalize_to_box(pdf):\n    x_min = pdf[[f\"x_right_hand_{i}\" for i in range (0,21)]].min().min()\n    x_max = pdf[[f\"x_right_hand_{i}\" for i in range (0,21)]].max().max()\n    y_min = pdf[[f\"y_right_hand_{i}\" for i in range (0,21)]].min().min()\n    y_max = pdf[[f\"y_right_hand_{i}\" for i in range (0,21)]].max().max()\n    \n    ratio = min(1/(x_max-x_min), 1/(y_max-y_min))\n    tmp = pdf.copy()\n    for i in range (0,21):\n        tmp[f\"x_right_hand_{i}\"] -= x_min\n        tmp[f\"x_right_hand_{i}\"] *= ratio\n        tmp[f\"y_right_hand_{i}\"] -= y_min\n        tmp[f\"y_right_hand_{i}\"] *= ratio\n        tmp[f\"z_right_hand_{i}\"] *= ratio\n    return tmp","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:24:56.899481Z","iopub.execute_input":"2026-02-23T23:24:56.900275Z","iopub.status.idle":"2026-02-23T23:24:56.917985Z","shell.execute_reply.started":"2026-02-23T23:24:56.900227Z","shell.execute_reply":"2026-02-23T23:24:56.917021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"row = train_df_split.iloc[20]\nX_pdf = read_right_hand_sequence_df(row[\"file_id\"], row[\"sequence_id\"])\nX_pdf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:24:56.919249Z","iopub.execute_input":"2026-02-23T23:24:56.920044Z","iopub.status.idle":"2026-02-23T23:24:57.145964Z","shell.execute_reply.started":"2026-02-23T23:24:56.920005Z","shell.execute_reply":"2026-02-23T23:24:57.144804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# distance between hand landmark 0 and 9 should be normalized, doesn't seem to be the case here, ToDo ?\n#(cf. https://storage.googleapis.com/mediapipe-assets/Model%20Card%20Hand%20Tracking%20(Lite_Full)%20with%20Fairness%20Oct%202021.pdf)\n\ndef dist(df, i,j):\n    return np.linalg.norm(df[[f\"{ax}_right_hand_{i}\" for ax in ['x', 'y', 'z']]].values - df[[f\"{ax}_right_hand_{j}\" for ax in ['x', 'y', 'z']]].values, axis=1)\n\n#X_pd[~X_pd.x_right_hand_0.isna()][['x_right_hand_0', 'y_right_hand_0', 'z_right_hand_0']]\nsns.displot(dist(X_pdf[~X_pdf['x_right_hand_0'].isna()], 0, 9))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:25:31.890612Z","iopub.execute_input":"2026-02-23T23:25:31.890984Z","iopub.status.idle":"2026-02-23T23:25:32.145061Z","shell.execute_reply.started":"2026-02-23T23:25:31.890954Z","shell.execute_reply":"2026-02-23T23:25:32.143799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_centered = center_wrist(X_pdf)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:25:38.425309Z","iopub.execute_input":"2026-02-23T23:25:38.426083Z","iopub.status.idle":"2026-02-23T23:25:38.449456Z","shell.execute_reply.started":"2026-02-23T23:25:38.426048Z","shell.execute_reply":"2026-02-23T23:25:38.448405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.displot(X_pdf[[f\"x_right_hand_{i}\" for i in [4, 8, 12, 16, 20]]], kind='kde')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:25:38.451135Z","iopub.execute_input":"2026-02-23T23:25:38.451503Z","iopub.status.idle":"2026-02-23T23:25:38.891948Z","shell.execute_reply.started":"2026-02-23T23:25:38.45146Z","shell.execute_reply":"2026-02-23T23:25:38.890791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.displot(X_centered[[f\"x_right_hand_{i}\" for i in [4, 8, 12, 16, 20]]], kind='kde')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:25:38.893753Z","iopub.execute_input":"2026-02-23T23:25:38.894139Z","iopub.status.idle":"2026-02-23T23:25:39.341177Z","shell.execute_reply.started":"2026-02-23T23:25:38.894107Z","shell.execute_reply":"2026-02-23T23:25:39.340249Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Visualization","metadata":{}},{"cell_type":"code","source":"!pip install mediapipe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:25:39.342997Z","iopub.execute_input":"2026-02-23T23:25:39.343335Z","iopub.status.idle":"2026-02-23T23:25:43.711833Z","shell.execute_reply.started":"2026-02-23T23:25:39.34328Z","shell.execute_reply":"2026-02-23T23:25:43.710361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_hands(seq_df):\n    import mediapipe as mp\n    from mediapipe.tasks.python.components.containers.landmark import NormalizedLandmark, Landmark\n    #import numpy as np\n    \n    mp_hands = mp.tasks.vision.HandLandmarksConnections\n    mp_drawing = mp.tasks.vision.drawing_utils\n    mp_drawing_styles = mp.tasks.vision.drawing_styles\n    \n    images = []\n    all_hand_landmarks = []\n    for seq_idx in range(len(seq_df)):\n        x_pose = seq_df.iloc[seq_idx].filter(regex=\"x_right_hand.*\").values\n        y_pose = seq_df.iloc[seq_idx].filter(regex=\"y_right_hand.*\").values\n        z_pose = seq_df.iloc[seq_idx].filter(regex=\"z_right_hand.*\").values\n\n        right_hand_landmarks = [NormalizedLandmark(x, y, z) for x, y, z in zip(x_pose, y_pose, z_pose)]\n\n        right_hand_image = np.zeros((900, 600, 3), dtype=\"int8\")\n\n        mp_drawing.draw_landmarks(\n              right_hand_image,\n              right_hand_landmarks,\n              mp_hands.HAND_CONNECTIONS,\n              mp_drawing_styles.get_default_hand_landmarks_style(),\n              mp_drawing_styles.get_default_hand_connections_style())\n        \n        x_pose = seq_df.iloc[seq_idx].filter(regex=\"x_left_hand.*\").values\n        y_pose = seq_df.iloc[seq_idx].filter(regex=\"y_left_hand.*\").values\n        z_pose = seq_df.iloc[seq_idx].filter(regex=\"z_left_hand.*\").values\n\n        left_hand_landmarks = [NormalizedLandmark(x, y, z) for x, y, z in zip(x_pose, y_pose, z_pose)]\n        \n        left_hand_image = np.zeros((900, 600, 3), dtype=\"int8\")\n\n        mp_drawing.draw_landmarks(\n                left_hand_image,\n                left_hand_landmarks,\n                mp_hands.HAND_CONNECTIONS,\n                mp_drawing_styles.get_default_hand_landmarks_style(),\n                mp_drawing_styles.get_default_hand_connections_style())\n        \n        images.append([right_hand_image.astype(np.uint8), left_hand_image.astype(np.uint8)])\n        all_hand_landmarks.append([right_hand_landmarks, left_hand_landmarks])\n    return images, all_hand_landmarks","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:25:43.713852Z","iopub.execute_input":"2026-02-23T23:25:43.714606Z","iopub.status.idle":"2026-02-23T23:25:43.728529Z","shell.execute_reply.started":"2026-02-23T23:25:43.714526Z","shell.execute_reply":"2026-02-23T23:25:43.727225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from matplotlib import animation, rc\nrc('animation', html='jshtml')\n\ndef create_animation(images):\n    fig = plt.figure(figsize=(2, 3))\n    ax = plt.Axes(fig, [0., 0., 1., 1.])\n    ax.set_axis_off()\n    fig.add_axes(ax)\n    im=ax.imshow(images[0], cmap=\"gray\")\n    plt.close(fig)\n    \n    def animate_func(i):\n        im.set_array(images[i])\n        return [im]\n\n    return animation.FuncAnimation(fig, animate_func, frames=len(images), interval=1000/10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:25:43.729796Z","iopub.execute_input":"2026-02-23T23:25:43.730091Z","iopub.status.idle":"2026-02-23T23:25:43.755445Z","shell.execute_reply.started":"2026-02-23T23:25:43.730063Z","shell.execute_reply":"2026-02-23T23:25:43.754522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hand_images, hand_landmarks = get_hands(X_pdf)\ncreate_animation(np.array(hand_images)[:, 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:25:43.75661Z","iopub.execute_input":"2026-02-23T23:25:43.756951Z","iopub.status.idle":"2026-02-23T23:25:56.228306Z","shell.execute_reply.started":"2026-02-23T23:25:43.756922Z","shell.execute_reply":"2026-02-23T23:25:56.227246Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hand_images, hand_landmarks = get_hands(normalize_to_box(X_pdf))\ncreate_animation(np.array(hand_images)[:, 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:25:56.229677Z","iopub.execute_input":"2026-02-23T23:25:56.23049Z","iopub.status.idle":"2026-02-23T23:26:04.74612Z","shell.execute_reply.started":"2026-02-23T23:25:56.23043Z","shell.execute_reply":"2026-02-23T23:26:04.74507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hand_images, hand_landmarks = get_hands(normalize_to_box(X_centered))\ncreate_animation(np.array(hand_images)[:, 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:26:04.747727Z","iopub.execute_input":"2026-02-23T23:26:04.748593Z","iopub.status.idle":"2026-02-23T23:26:12.648247Z","shell.execute_reply.started":"2026-02-23T23:26:04.748548Z","shell.execute_reply":"2026-02-23T23:26:12.647239Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Normalize sequence length","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def normalize_frames(X, max_frames=MAX_FRAMES):\n    T, D = X.shape\n\n    if T > max_frames:\n        return X[:max_frames]\n\n    if T < max_frames:\n        pad = np.zeros((max_frames - T, D), dtype=np.float32)\n        return np.vstack([X, pad])\n\n    return X","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:26:12.652842Z","iopub.execute_input":"2026-02-23T23:26:12.653289Z","iopub.status.idle":"2026-02-23T23:26:12.659501Z","shell.execute_reply.started":"2026-02-23T23:26:12.653244Z","shell.execute_reply":"2026-02-23T23:26:12.658496Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#row = train_df_split.iloc[20]\n\n#X = read_right_hand_sequence(row[\"file_id\"], row[\"sequence_id\"])\nX = X_centered.values.astype(np.float32)\nXn = normalize_frames(X)\n\nprint(\"Original:\", X.shape)\nprint(\"Normalizado:\", Xn.shape)\nprint(Xn[:2])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:26:12.660734Z","iopub.execute_input":"2026-02-23T23:26:12.661063Z","iopub.status.idle":"2026-02-23T23:26:12.680282Z","shell.execute_reply.started":"2026-02-23T23:26:12.661024Z","shell.execute_reply":"2026-02-23T23:26:12.679248Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"row = train_df_split.iloc[0]\nfile_id, sequence_id = row[\"file_id\"], row[\"sequence_id\"]\n\npath = f\"/kaggle/input/asl-fingerspelling/train_landmarks/{file_id}.parquet\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:26:12.681782Z","iopub.execute_input":"2026-02-23T23:26:12.682104Z","iopub.status.idle":"2026-02-23T23:26:12.697482Z","shell.execute_reply.started":"2026-02-23T23:26:12.682075Z","shell.execute_reply":"2026-02-23T23:26:12.696391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pq_file = pq.ParquetFile(path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:26:12.698869Z","iopub.execute_input":"2026-02-23T23:26:12.699275Z","iopub.status.idle":"2026-02-23T23:26:12.737708Z","shell.execute_reply.started":"2026-02-23T23:26:12.699234Z","shell.execute_reply":"2026-02-23T23:26:12.736756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def count_valid_frames(X):\n    # Un frame es válido si NO todo es NaN\n    return np.sum(~np.all(np.isnan(X), axis=1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:26:12.739287Z","iopub.execute_input":"2026-02-23T23:26:12.739778Z","iopub.status.idle":"2026-02-23T23:26:12.745945Z","shell.execute_reply.started":"2026-02-23T23:26:12.73972Z","shell.execute_reply":"2026-02-23T23:26:12.744692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nclass ASLRightHandDataset(Dataset):\n    def __init__(self, df, max_frames=160):\n        self.df = df.reset_index(drop=True)\n        self.max_frames = max_frames\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n\n        X_raw = read_right_hand_sequence_df(\n            row[\"file_id\"],\n            row[\"sequence_id\"]\n        )  \n        X_centered = center_wrist(X_raw)\n    \n        input_len = count_valid_frames(X_centered)\n    \n        Y = torch.tensor(row[\"encoded\"], dtype=torch.long)\n        target_len = len(Y)\n\n        if input_len < target_len:\n            return None\n\n\n        X = normalize_frames(X_centered, self.max_frames)\n    \n        X = np.nan_to_num(X, nan=0.0)\n    \n        X = torch.tensor(X, dtype=torch.float32)\n    \n        input_len = min(input_len, self.max_frames)\n    \n        return X, Y, input_len, target_len\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:26:12.746906Z","iopub.execute_input":"2026-02-23T23:26:12.747202Z","iopub.status.idle":"2026-02-23T23:26:12.760757Z","shell.execute_reply.started":"2026-02-23T23:26:12.747173Z","shell.execute_reply":"2026-02-23T23:26:12.75973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def collate_fn(batch):\n    # Quitamos ejemplos inválidos (None)\n    batch = [b for b in batch if b is not None]\n\n    # Si todo el batch era inválido, devolvemos None\n    if len(batch) == 0:\n        return None\n\n    Xs = []\n    Ys = []\n    in_lens = []\n    tar_lens = []\n\n    for X, Y, in_len, tar_len in batch:\n        Xs.append(X)\n        Ys.append(Y)\n        in_lens.append(in_len)\n        tar_lens.append(tar_len)\n\n    Xs = torch.stack(Xs)\n    Ys = torch.cat(Ys)\n    in_lens = torch.tensor(in_lens, dtype=torch.long)\n    tar_lens = torch.tensor(tar_lens, dtype=torch.long)\n\n    return Xs, Ys, in_lens, tar_lens\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:26:12.762135Z","iopub.execute_input":"2026-02-23T23:26:12.762544Z","iopub.status.idle":"2026-02-23T23:26:12.786998Z","shell.execute_reply.started":"2026-02-23T23:26:12.762514Z","shell.execute_reply":"2026-02-23T23:26:12.785914Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df_tiny = train_df_split.sample(200, random_state=0)\ntrain_dataset_tiny = ASLRightHandDataset(train_df_tiny)\n\ntrain_loader_tiny = DataLoader(\n    train_dataset_tiny,\n    batch_size=4,\n    shuffle=True,\n    collate_fn=collate_fn,\n    num_workers=0\n)\n\nX, Y, in_len, tar_len = next(iter(train_loader_tiny))\n\nprint(\"X:\", X.shape)          # (B, 160, 63)\nprint(\"Y:\", Y.shape)          # (sum of target lengths)\nprint(\"in_len:\", in_len)\nprint(\"tar_len:\", tar_len)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:26:12.78839Z","iopub.execute_input":"2026-02-23T23:26:12.788799Z","iopub.status.idle":"2026-02-23T23:26:13.880939Z","shell.execute_reply.started":"2026-02-23T23:26:12.788758Z","shell.execute_reply":"2026-02-23T23:26:13.879899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nclass SimpleRNN(nn.Module):\n    def __init__(self, input_size, hidden_size, num_classes):\n        super().__init__()\n\n        self.rnn = nn.RNN(\n            input_size=input_size,\n            hidden_size=hidden_size,\n            batch_first=True\n        )\n\n        self.classifier = nn.Linear(hidden_size, num_classes)\n        self.log_softmax = nn.LogSoftmax(dim=-1)\n\n    def forward(self, x):\n        out, _ = self.rnn(x)        # (B, T, H)\n        out = self.classifier(out)  # (B, T, C)\n        out = self.log_softmax(out)\n        out = out.permute(1, 0, 2)  # (T, B, C) para CTC\n        return out\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:26:13.882383Z","iopub.execute_input":"2026-02-23T23:26:13.882698Z","iopub.status.idle":"2026-02-23T23:26:13.890029Z","shell.execute_reply.started":"2026-02-23T23:26:13.88267Z","shell.execute_reply":"2026-02-23T23:26:13.889083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class EmbeddedRNN(nn.Module):\n    def __init__(self, input_size, embed_size, hidden_size, num_classes):\n        super().__init__()\n\n        self.embedding = nn.Linear(input_size, embed_size)\n\n        self.rnn = nn.RNN(\n            input_size=embed_size,\n            hidden_size=hidden_size,\n            batch_first=True\n        )\n\n        self.classifier = nn.Linear(hidden_size, num_classes)\n        self.log_softmax = nn.LogSoftmax(dim=-1)\n\n    def forward(self, x):\n\n        x = self.embedding(x)       \n        out, _ = self.rnn(x)      \n        out = self.classifier(out)  \n        out = self.log_softmax(out)\n        out = out.permute(1, 0, 2)  \n\n        return out","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:26:13.891436Z","iopub.execute_input":"2026-02-23T23:26:13.891857Z","iopub.status.idle":"2026-02-23T23:26:13.916408Z","shell.execute_reply.started":"2026-02-23T23:26:13.891816Z","shell.execute_reply":"2026-02-23T23:26:13.915016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_classes = len(letter_to_int)\n\n\nmodel = SimpleRNN(\n    input_size=63,\n    hidden_size=128,\n    num_classes=num_classes\n).to(DEVICE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:26:13.918062Z","iopub.execute_input":"2026-02-23T23:26:13.918596Z","iopub.status.idle":"2026-02-23T23:26:13.941421Z","shell.execute_reply.started":"2026-02-23T23:26:13.918546Z","shell.execute_reply":"2026-02-23T23:26:13.940596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = EmbeddedRNN(\n    input_size=63,         \n    embed_size=128,         \n    hidden_size=128,\n    num_classes=len(letter_to_int)\n).to(DEVICE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:26:13.942621Z","iopub.execute_input":"2026-02-23T23:26:13.942921Z","iopub.status.idle":"2026-02-23T23:26:13.961156Z","shell.execute_reply.started":"2026-02-23T23:26:13.942885Z","shell.execute_reply":"2026-02-23T23:26:13.959878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.eval()\n\nwith torch.no_grad():\n    outputs = model(X.to(DEVICE))\n\nprint(outputs.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:26:13.962862Z","iopub.execute_input":"2026-02-23T23:26:13.963577Z","iopub.status.idle":"2026-02-23T23:26:14.034631Z","shell.execute_reply.started":"2026-02-23T23:26:13.963541Z","shell.execute_reply":"2026-02-23T23:26:14.033612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ncriterion = nn.CTCLoss(\n    blank=letter_to_int[\"<blank>\"],\n    zero_infinity=True\n)\n\nloss = criterion(\n    outputs,    # (T, B, C)\n    Y.to(DEVICE),\n    in_len,\n    tar_len\n)\n\nprint(\"CTC loss:\", loss.item())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:26:14.035693Z","iopub.execute_input":"2026-02-23T23:26:14.035982Z","iopub.status.idle":"2026-02-23T23:26:14.058179Z","shell.execute_reply.started":"2026-02-23T23:26:14.035954Z","shell.execute_reply":"2026-02-23T23:26:14.057204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_one_epoch(model, loader, optimizer, criterion, device):\n    model.train()\n    total_loss = 0.0\n    num_batches = 0\n\n    for batch in loader:\n\n        if batch is None:\n            continue\n\n        X, Y, input_lens, target_lens = batch\n\n        X = X.to(device)\n        Y = Y.to(device)\n        input_lens = input_lens.to(device)\n        target_lens = target_lens.to(device)\n\n        optimizer.zero_grad()\n\n        outputs = model(X)  # (T, B, C)\n\n        loss = criterion(\n            outputs,\n            Y,\n            input_lens,\n            target_lens\n        )\n\n        loss.backward()\n\n\n        optimizer.step()\n\n        total_loss += loss.item()\n        num_batches += 1\n\n    avg_loss = total_loss / num_batches\n\n    return avg_loss\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:26:14.059614Z","iopub.execute_input":"2026-02-23T23:26:14.059951Z","iopub.status.idle":"2026-02-23T23:26:14.067559Z","shell.execute_reply.started":"2026-02-23T23:26:14.059921Z","shell.execute_reply":"2026-02-23T23:26:14.066295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_epochs = 3\ntrain_losses = []\ngrad_norms = []\noptimizer = optim.Adam(model.parameters(), lr=1e-3)\n\n\nfor epoch in range(num_epochs):\n    loss = train_one_epoch(\n        model,\n        train_loader_tiny,\n        optimizer,\n        criterion,\n        DEVICE\n    )\n    train_losses.append(loss)\n    print(f\"Epoch {epoch+1}: loss = {loss:.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:26:14.069483Z","iopub.execute_input":"2026-02-23T23:26:14.069848Z","iopub.status.idle":"2026-02-23T23:27:54.885601Z","shell.execute_reply.started":"2026-02-23T23:26:14.069818Z","shell.execute_reply":"2026-02-23T23:27:54.884239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(6,4))\nplt.plot(train_losses, marker=\"o\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"CTC Loss\")\nplt.title(\"Training Loss over Epochs\")\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T23:27:54.886966Z","iopub.execute_input":"2026-02-23T23:27:54.887763Z","iopub.status.idle":"2026-02-23T23:27:55.059238Z","shell.execute_reply.started":"2026-02-23T23:27:54.887726Z","shell.execute_reply":"2026-02-23T23:27:55.058336Z"}},"outputs":[],"execution_count":null}]}