{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":52950,"databundleVersionId":5973250,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\nos.environ['TF_XLA_FLAGS'] = '--tf_xla_enable_xla_devices'\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\n\n# Disable XLA JIT to avoid CTC compatibility issues\ntf.config.optimizer.set_jit(False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T21:34:06.869676Z","iopub.execute_input":"2025-06-26T21:34:06.869935Z","iopub.status.idle":"2025-06-26T21:34:22.199848Z","shell.execute_reply.started":"2025-06-26T21:34:06.869917Z","shell.execute_reply":"2025-06-26T21:34:22.199247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    SEQ_LENGTH = 384       \n    LANDMARK_SIZE = 84     \n    BATCH_SIZE = 64\n    EPOCHS = 15\n    CHAR_MAX_LEN = 32      \n    RATE = 0.2             \n    EMBED_DIM = 256        \n    \nconfig = Config()\n\n\ndef get_landmark_cols():\n    cols = []\n    for coord in ['x', 'y']:\n        for hand in ['left_hand', 'right_hand']:\n            for i in range(21):\n                cols.append(f'{coord}_{hand}_{i}')\n    return cols\n\n\nLANDMARK_COLS = get_landmark_cols()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T21:34:22.200472Z","iopub.execute_input":"2025-06-26T21:34:22.200887Z","iopub.status.idle":"2025-06-26T21:34:22.205591Z","shell.execute_reply.started":"2025-06-26T21:34:22.200869Z","shell.execute_reply":"2025-06-26T21:34:22.204966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ROOT_PATH = \"/kaggle/input/asl-fingerspelling\"\ntrain_df = pd.read_csv(f\"{ROOT_PATH}/train.csv\")\nchar_to_num = keras.layers.StringLookup(vocabulary=list(\"abcdefghijklmnopqrstuvwxyz' !\"), oov_token=\"\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T21:34:22.207354Z","iopub.execute_input":"2025-06-26T21:34:22.207546Z","iopub.status.idle":"2025-06-26T21:34:23.62432Z","shell.execute_reply.started":"2025-06-26T21:34:22.207531Z","shell.execute_reply":"2025-06-26T21:34:23.623479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_and_preprocess_sequence(file_id, sequence_id):\n    \"\"\"Load and preprocess individual sequence\"\"\"\n    file_path = f\"{ROOT_PATH}/train_landmarks/{file_id}.parquet\"\n    df = pd.read_parquet(file_path)\n    \n    # Filter by sequence_id\n    seq_df = df[df.index == sequence_id]\n    \n    # Handle missing sequences\n    if seq_df.empty:\n        return np.zeros((config.SEQ_LENGTH, config.LANDMARK_SIZE))\n    \n    # Extract relevant landmarks\n    seq_data = seq_df[LANDMARK_COLS].fillna(0).values.astype(np.float32)\n    n_frames = seq_data.shape[0]\n    \n    # Pad or truncate to fixed length\n    if n_frames < config.SEQ_LENGTH:\n        pad_len = config.SEQ_LENGTH - n_frames\n        seq_data = np.pad(seq_data, ((0, pad_len), (0, 0)), mode='constant')\n    else:\n        seq_data = seq_data[:config.SEQ_LENGTH]\n    \n    return seq_data\n\n\n# Preprocess subset of data\nsubset_df = train_df.head(1000).copy()\nX = []\ny = []\n\nfor _, row in tqdm(subset_df.iterrows(), total=len(subset_df), desc=\"Loading and preprocess sequences\"):\n    seq_data = load_and_preprocess_sequence(row['file_id'], row['sequence_id'])\n    X.append(seq_data)\n    y.append(row['phrase'])\n\nX = np.array(X)\ny = np.array(y)\n\n# Encode labels\ny_encoded = char_to_num(tf.strings.unicode_split(y, input_encoding=\"UTF-8\"))\n# Convert RaggedTensor to padded tensor\ny_padded = y_encoded.to_tensor(default_value=0)\n# Truncate sequences longer than max length\ny_padded = y_padded[:, :config.CHAR_MAX_LEN]\n# Pad to fixed length\npad_len = config.CHAR_MAX_LEN - tf.shape(y_padded)[1]\ny_padded = tf.pad(y_padded, [[0, 0], [0, pad_len]], constant_values=0)\n\n# Train-validation split\nX_train, X_val, y_train, y_val = train_test_split(X, y_padded.numpy(), test_size=0.1, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T21:34:23.625133Z","iopub.execute_input":"2025-06-26T21:34:23.625434Z","iopub.status.idle":"2025-06-26T22:05:02.650673Z","shell.execute_reply.started":"2025-06-26T21:34:23.625411Z","shell.execute_reply":"2025-06-26T22:05:02.649957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_model():\n    inputs = keras.Input(shape=(config.SEQ_LENGTH, config.LANDMARK_SIZE))\n    \n    # Encoder\n    x = layers.Dense(config.EMBED_DIM, activation=\"relu\")(inputs)\n    x = layers.LayerNormalization()(x)\n    x = layers.Dropout(config.RATE)(x)\n    \n    # Transformer Blocks\n    for _ in range(2):\n        # Self-attention\n        attn = layers.MultiHeadAttention(num_heads=4, key_dim=128)(x, x)\n        attn = layers.Dropout(config.RATE)(attn)\n        x = layers.Add()([x, attn])\n        x = layers.LayerNormalization()(x)\n        \n        # Feed-forward network\n        ffn = layers.Dense(4 * config.EMBED_DIM, activation=\"relu\")(x)\n        ffn = layers.Dense(config.EMBED_DIM)(ffn)  # Maintain dimension\n        ffn = layers.Dropout(config.RATE)(ffn)\n        x = layers.Add()([x, ffn])\n        x = layers.LayerNormalization()(x)\n    \n    # CTC requires time-distributed outputs\n    x = layers.Dense(128, activation=\"relu\")(x)\n    outputs = layers.Dense(len(char_to_num.get_vocabulary()) + 1)(x)  # +1 for CTC blank\n    \n    model = keras.Model(inputs=inputs, outputs=outputs)\n    return model\n\n\nmodel = build_model()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T22:05:02.651457Z","iopub.execute_input":"2025-06-26T22:05:02.651701Z","iopub.status.idle":"2025-06-26T22:05:03.980244Z","shell.execute_reply.started":"2025-06-26T22:05:02.651684Z","shell.execute_reply":"2025-06-26T22:05:03.979709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def ctc_loss(y_true, y_pred):\n    # Calculate label lengths (number of non-zero characters)\n    mask = tf.not_equal(y_true, 0)\n    label_length = tf.reduce_sum(tf.cast(mask, tf.int32), axis=-1)\n    \n    # Calculate input lengths (all sequences are full length)\n    batch_size = tf.shape(y_true)[0]\n    input_length = tf.fill([batch_size], tf.shape(y_pred)[1])\n    \n    # Use TensorFlow's native CTC loss\n    loss = tf.nn.ctc_loss(\n        labels=tf.cast(y_true, tf.int32),\n        logits=y_pred,\n        label_length=label_length,\n        logit_length=input_length,\n        logits_time_major=False\n    )\n    return tf.reduce_mean(loss)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T22:05:03.980919Z","iopub.execute_input":"2025-06-26T22:05:03.981099Z","iopub.status.idle":"2025-06-26T22:05:03.985667Z","shell.execute_reply.started":"2025-06-26T22:05:03.981085Z","shell.execute_reply":"2025-06-26T22:05:03.984994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(\n    optimizer=keras.optimizers.Adam(learning_rate=0.001),\n    loss=ctc_loss,\n    metrics=[]\n)\n\nearly_stopping = keras.callbacks.EarlyStopping(\n    monitor=\"val_loss\", patience=3, restore_best_weights=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T22:05:03.986397Z","iopub.execute_input":"2025-06-26T22:05:03.986646Z","iopub.status.idle":"2025-06-26T22:05:04.010917Z","shell.execute_reply.started":"2025-06-26T22:05:03.98663Z","shell.execute_reply":"2025-06-26T22:05:04.010316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(\n    X_train,\n    y_train,\n    validation_data=(X_val, y_val),\n    epochs=config.EPOCHS,\n    batch_size=config.BATCH_SIZE,\n    callbacks=[early_stopping],\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T22:05:04.01165Z","iopub.execute_input":"2025-06-26T22:05:04.012362Z","iopub.status.idle":"2025-06-26T22:05:46.672278Z","shell.execute_reply.started":"2025-06-26T22:05:04.012334Z","shell.execute_reply":"2025-06-26T22:05:46.671649Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def decode_pred(pred):\n    \"\"\"Convert model output to text\"\"\"\n    input_len = np.ones(pred.shape[0]) * pred.shape[1]\n    results = keras.backend.ctc_decode(\n        pred, \n        input_length=input_len, \n        greedy=True\n    )[0][0]\n    texts = tf.strings.reduce_join(char_to_num(results), axis=-1)\n    return [str(t.numpy(), \"utf-8\") for t in texts]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T22:05:46.674874Z","iopub.execute_input":"2025-06-26T22:05:46.675099Z","iopub.status.idle":"2025-06-26T22:05:46.679944Z","shell.execute_reply.started":"2025-06-26T22:05:46.675082Z","shell.execute_reply":"2025-06-26T22:05:46.679288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check if test data exists\nTEST_CSV_PATH = f\"{ROOT_PATH}/test.csv\"\n\nif os.path.exists(TEST_CSV_PATH):\n    # Load test data\n    test_df = pd.read_csv(TEST_CSV_PATH)\n    X_test = []\n\n    for _, row in test_df.iterrows():\n        seq_data = load_and_preprocess_sequence(row['file_id'], row['sequence_id'])\n        X_test.append(seq_data)\n\n    X_test = np.array(X_test)\n\n    # Predict\n    predictions = model.predict(X_test)\n    decoded_preds = decode_pred(predictions)\n\n    # Create submission\n    submission = pd.DataFrame({\n        \"sequence_id\": test_df.sequence_id,\n        \"phrase\": decoded_preds\n    })\n    submission.to_csv(\"submission.csv\", index=False)\n    print(\"Submission file created with test predictions!\")\nelse:\n    # Create a sample submission file for local testing\n    print(\"Test data not found. Creating sample submission file.\")\n    sample_submission = pd.DataFrame({\n        \"sequence_id\": [1, 2, 3],\n        \"phrase\": [\"hello\", \"world\", \"asl\"]\n    })\n    sample_submission.to_csv(\"submission.csv\", index=False)\n    print(\"Sample submission file created for testing.\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-26T22:05:46.680562Z","iopub.execute_input":"2025-06-26T22:05:46.680768Z","iopub.status.idle":"2025-06-26T22:05:46.706452Z","shell.execute_reply.started":"2025-06-26T22:05:46.680729Z","shell.execute_reply":"2025-06-26T22:05:46.705795Z"}},"outputs":[],"execution_count":null}]}