{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":52950,"databundleVersionId":5973250,"sourceType":"competition"},{"sourceId":12532057,"sourceType":"datasetVersion","datasetId":7911157},{"sourceId":12595184,"sourceType":"datasetVersion","datasetId":7955257},{"sourceId":12610606,"sourceType":"datasetVersion","datasetId":7965957}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# =================================================================================\n# CELL 1: SETUP AND IMPORTS\n# =================================================================================\n!pip install -q rapidfuzz scikit-learn\n\nimport os\nimport json\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader, RandomSampler, SequentialSampler\nfrom sklearn.model_selection import train_test_split\nfrom tqdm.auto import tqdm\nfrom types import SimpleNamespace\nimport torch.multiprocessing as mp\nfrom torch.utils.data.distributed import DistributedSampler\nfrom torch.nn.parallel import DistributedDataParallel as DDP\nimport torch.distributed as dist","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-29T12:47:13.175641Z","iopub.execute_input":"2025-07-29T12:47:13.17591Z","iopub.status.idle":"2025-07-29T12:47:25.303914Z","shell.execute_reply.started":"2025-07-29T12:47:13.175887Z","shell.execute_reply":"2025-07-29T12:47:25.303333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =================================================================================\n# CELL 2: MODEL ARCHITECTURE\n# =================================================================================\nclass ECA(nn.Module):\n    def __init__(self, kernel_size=5):\n        super(ECA, self).__init__()\n        self.conv = nn.Conv1d(1, 1, kernel_size=kernel_size, padding='same', bias=False)\n    def forward(self, x):\n        x_permuted = x.permute(0, 2, 1)\n        nn_ = F.adaptive_avg_pool1d(x_permuted, 1)\n        nn_ = self.conv(nn_.transpose(-1, -2)).transpose(-1, -2)\n        nn_ = torch.sigmoid(nn_)\n        return x * nn_.permute(0, 2, 1)\n\nclass Conv1DBlock(nn.Module):\n    def __init__(self, channel_size, kernel_size, drop_rate=0.0, expand_ratio=2, strides=1):\n        super(Conv1DBlock, self).__init__()\n        self.strides, self.kernel_size = strides, kernel_size\n        channels_expand = channel_size * expand_ratio\n        self.pre_bn = nn.BatchNorm1d(channel_size)\n        self.expand_conv = nn.Linear(channel_size, channels_expand, bias=True)\n        self.dw_conv = nn.Conv1d(channels_expand, channels_expand, kernel_size=kernel_size,\n                                 stride=strides, padding=0, groups=channels_expand, bias=False)\n        self.conv_bn = nn.BatchNorm1d(channels_expand)\n        self.eca = ECA()\n        self.project_conv = nn.Linear(channels_expand, channel_size, bias=True)\n        self.drop = nn.Dropout(drop_rate)\n    def forward(self, x):\n        skip = x\n        x = self.pre_bn(x.permute(0, 2, 1)).permute(0, 2, 1)\n        x = F.silu(self.expand_conv(x))\n        if self.strides > 1:\n            total_pad = self.kernel_size - self.strides\n            pad_left, pad_right = total_pad // 2, total_pad - (total_pad // 2)\n            x = F.pad(x.permute(0, 2, 1), (pad_left, pad_right), \"constant\", 0).permute(0, 2, 1)\n        else:\n            x = F.pad(x.permute(0, 2, 1), (self.kernel_size // 2, self.kernel_size // 2), \"constant\", 0).permute(0, 2, 1)\n        x = self.dw_conv(x.permute(0, 2, 1)).permute(0, 2, 1)\n        x = self.conv_bn(x.permute(0, 2, 1)).permute(0, 2, 1)\n        x = self.eca(x)\n        x = self.project_conv(x)\n        x = self.drop(x)\n        if self.strides == 1: x = x + skip\n        return x\n\nclass MultiHeadSelfAttention(nn.Module):\n    def __init__(self, dim=256, num_heads=4, dropout=0.0):\n        super(MultiHeadSelfAttention, self).__init__()\n        self.num_heads, self.head_dim = num_heads, dim // num_heads\n        self.scale = self.head_dim ** -0.5\n        self.qkv = nn.Linear(dim, dim * 3, bias=False)\n        self.drop1 = nn.Dropout(dropout)\n        self.proj = nn.Linear(dim, dim, bias=False)\n    def forward(self, x):\n        B, T, C = x.shape\n        qkv = self.qkv(x).reshape(B, T, 3, self.num_heads, self.head_dim).permute(2, 0, 3, 1, 4)\n        q, k, v = qkv[0], qkv[1], qkv[2]\n        attn = (q @ k.transpose(-2, -1)) * self.scale\n        attn = attn.softmax(dim=-1)\n        attn = self.drop1(attn)\n        x = (attn @ v).transpose(1, 2).reshape(B, T, C)\n        return self.proj(x)\n\nclass TransformerBlock(nn.Module):\n    def __init__(self, dim=256, num_heads=4, expand=4, attn_dropout=0.2, drop_rate=0.2):\n        super(TransformerBlock, self).__init__()\n        self.bn1 = nn.BatchNorm1d(dim)\n        self.mhsa = MultiHeadSelfAttention(dim, num_heads, attn_dropout)\n        self.drop1 = nn.Dropout(drop_rate)\n        self.bn2 = nn.BatchNorm1d(dim)\n        self.fc1 = nn.Linear(dim, dim * expand, bias=False)\n        self.fc2 = nn.Linear(dim * expand, dim, bias=False)\n        self.drop2 = nn.Dropout(drop_rate)\n    def forward(self, x):\n        skip1 = x\n        x = self.bn1(x.permute(0, 2, 1)).permute(0, 2, 1)\n        x = self.mhsa(x)\n        x = self.drop1(x)\n        x = x + skip1\n        skip2 = x\n        x = self.bn2(x.permute(0, 2, 1)).permute(0, 2, 1)\n        x = F.silu(self.fc1(x))\n        x = self.fc2(x)\n        x = self.drop2(x)\n        x = x + skip2\n        return x\n\nclass Encoder(nn.Module):\n    def __init__(self, in_channels=390, dim=192, drop_rate=0.2, ksize=17):\n        super(Encoder, self).__init__()\n        self.stem = nn.Linear(in_channels, dim, bias=False)\n        self.blocks = nn.Sequential(\n            Conv1DBlock(dim, ksize, expand_ratio=4, drop_rate=drop_rate),\n            TransformerBlock(dim, num_heads=4, expand=2, drop_rate=drop_rate, attn_dropout=0.2),\n            Conv1DBlock(dim, ksize, expand_ratio=4, drop_rate=drop_rate),\n            TransformerBlock(dim, num_heads=4, expand=2, drop_rate=drop_rate, attn_dropout=0.2),\n            Conv1DBlock(dim, ksize, expand_ratio=4, drop_rate=0, strides=2),\n            TransformerBlock(dim, num_heads=4, expand=2, drop_rate=drop_rate, attn_dropout=0.2),\n            Conv1DBlock(dim, ksize, expand_ratio=4, drop_rate=drop_rate),\n            TransformerBlock(dim, num_heads=4, expand=2, drop_rate=drop_rate, attn_dropout=0.2)\n        )\n        self.final_bn = nn.BatchNorm1d(dim)\n    def forward(self, x):\n        x = self.stem(x)\n        x = self.blocks(x)\n        return self.final_bn(x.permute(0, 2, 1)).permute(0, 2, 1)\n\nclass PosEmbedding(nn.Module):\n    def __init__(self, dim=192, max_len=128):\n        super().__init__()\n        self.pos_emb = nn.Embedding(max_len, dim)\n    def forward(self, x):\n        positions = torch.arange(x.size(1), device=x.device)\n        return x + self.pos_emb(positions)\n\nclass CTCDecoder(nn.Module):\n    def __init__(self, in_dim=192, out_dim=62):\n        super().__init__()\n        self.gru = nn.LSTM(in_dim, in_dim, batch_first=True)\n        self.fc1 = nn.Linear(in_dim, in_dim * 2)\n        self.dropout = nn.Dropout(0.5)\n        self.classifier = nn.Linear(in_dim * 2, out_dim)\n    def forward(self, x):\n        x, _ = self.gru(x)\n        x = self.fc1(x)\n        x = self.dropout(x)\n        return self.classifier(x)\n\nclass AttentionDecoder(nn.Module):\n    def __init__(self, dim=192, num_heads=4, out_dim=62, max_len=128):\n        super().__init__()\n        self.embedding = nn.Embedding(out_dim, dim, padding_idx=0)\n        self.pos_embedding = PosEmbedding(dim, max_len)\n        decoder_layer = nn.TransformerDecoderLayer(d_model=dim, nhead=num_heads, dim_feedforward=dim*2, dropout=0.5, batch_first=True)\n        self.decoder = nn.TransformerDecoder(decoder_layer, num_layers=1)\n        self.classifier = nn.Linear(dim, out_dim)\n    def forward(self, targets, encoder_output):\n        tgt_mask = nn.Transformer.generate_square_subsequent_mask(targets.size(1), device=targets.device)\n        tgt_padding_mask = (targets == 0)\n        tgt_embedded = self.pos_embedding(self.embedding(targets))\n        output = self.decoder(tgt=tgt_embedded, memory=encoder_output, tgt_mask=tgt_mask, tgt_key_padding_mask=tgt_padding_mask)\n        return self.classifier(output)\n\nclass FullModel(nn.Module):\n    def __init__(self, in_channels=390, dim=192, out_dim=62):\n        super().__init__()\n        self.encoder = Encoder(in_channels, dim)\n        self.ctc_decoder = CTCDecoder(dim, out_dim)\n        self.attention_decoder = AttentionDecoder(dim, 4, out_dim)\n    def forward(self, x, targets):\n        encoder_output = self.encoder(x)\n        ctc_logits = self.ctc_decoder(encoder_output)\n        attn_logits = self.attention_decoder(targets, encoder_output)\n        return ctc_logits, attn_logits","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-29T09:59:07.677183Z","iopub.execute_input":"2025-07-29T09:59:07.677433Z","iopub.status.idle":"2025-07-29T09:59:07.704411Z","shell.execute_reply.started":"2025-07-29T09:59:07.677408Z","shell.execute_reply":"2025-07-29T09:59:07.703718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =================================================================================\n# CELL 3: DATALOADER AND HELPERS\n# =================================================================================\nfrom rapidfuzz.distance.DamerauLevenshtein_py import distance as rapidfuzz_dist\n\nclass PreprocessedDataset(Dataset):\n    def __init__(self, df, cfg, use_augmentations=True):\n        self.df = df.reset_index(drop=True)\n        self.use_augmentations = use_augmentations\n        self.cfg = cfg\n        self.root_dir = '/kaggle/input/asl-final-preprocessed-tensors/'\n\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        \n        # Choose which relative path to use from the CSV\n        relative_path = row['path_aug'] if self.use_augmentations else row['path_no_aug']\n        \n        # --- NEW: Construct the full, absolute path ---\n        # This joins the root directory with the relative path from the CSV\n        full_path = os.path.join(self.root_dir, relative_path)\n        \n        # Load the final, model-ready tensor using the full path\n        input_tensor = torch.load(full_path)\n        \n        # Tokenize the phrase\n        phrase = str(row['phrase'])\n        token_ids, _ = self.tokenize(phrase)\n        \n        return {'input': input_tensor, 'token_ids': token_ids, 'phrase': phrase}\n\n    def tokenize(self, phrase):\n        char_to_num = self.cfg.tokenizer[0]\n        pad_token_id = char_to_num[self.cfg.pad_token]\n        end_token_id = char_to_num[self.cfg.end_token]\n        max_phrase = self.cfg.max_phrase\n        \n        phrase_ids = [char_to_num[char] for char in phrase if char in char_to_num]\n        if len(phrase_ids) > max_phrase - 1:\n            phrase_ids = phrase_ids[:max_phrase - 1]\n        phrase_ids = phrase_ids + [end_token_id]\n        \n        attention_mask = [1] * len(phrase_ids)\n        to_pad = max_phrase - len(phrase_ids)\n        phrase_ids = phrase_ids + [pad_token_id] * to_pad\n        attention_mask = attention_mask + [0] * to_pad\n        return torch.tensor(phrase_ids).long(), torch.tensor(attention_mask).long()\n\ndef get_score(phrase_gt, phrase_preds):\n    N = np.sum([len(p) for p in phrase_gt])\n    D = np.sum([rapidfuzz_dist(p1, p2) for p1, p2 in zip(phrase_gt, phrase_preds)])\n    return (N - D) / N if N > 0 else 0\n\ndef get_wer(phrase_gt, phrase_preds):\n    \"\"\"Calculates the Word Error Rate.\"\"\"\n    # Split phrases into words\n    words_gt = [p.split() for p in phrase_gt]\n    words_preds = [p.split() for p in phrase_preds]\n    \n    # Calculate Levenshtein distance at the word level\n    D = np.sum([rapidfuzz_dist(w1, w2) for w1, w2 in zip(words_gt, words_preds)])\n    # Count total number of words in the ground truth\n    N = np.sum([len(w) for w in words_gt])\n    \n    return D / N if N > 0 else 0\n\nclass DetailedCombinedLoss(nn.Module):\n    def __init__(self, num_classes=62):\n        super().__init__()\n        self.ctc_loss = nn.CTCLoss(blank=0, reduction='mean', zero_infinity=True)\n        self.attn_loss = nn.CrossEntropyLoss(ignore_index=0)\n    def forward(self, ctc_logits, attn_logits, targets, ctc_input_lengths, target_lengths):\n        ctc_log_probs = F.log_softmax(ctc_logits, dim=2).permute(1, 0, 2)\n        loss_ctc = self.ctc_loss(ctc_log_probs, targets, ctc_input_lengths, target_lengths)\n        loss_attn = self.attn_loss(attn_logits.reshape(-1, attn_logits.size(-1)), targets.reshape(-1))\n        return 0.75 * loss_attn + 0.25 * loss_ctc, loss_attn, loss_ctc\n\nclass EarlyStopping:\n    \"\"\"Stops training when a monitored metric has stopped improving.\"\"\"\n    def __init__(self, patience=10, min_delta=0, verbose=False):\n        \"\"\"\n        Args:\n            patience (int): How long to wait after last time validation score improved.\n            min_delta (float): Minimum change in the monitored quantity to qualify as an improvement.\n            verbose (bool): If True, prints a message for each validation score improvement.\n        \"\"\"\n        self.patience = patience\n        self.min_delta = min_delta\n        self.verbose = verbose\n        self.counter = 0\n        self.best_score = None\n        self.early_stop = False\n\n    def __call__(self, val_score, model):\n        # NOTE: This callback is designed for a metric where a HIGHER score is BETTER (like Levenshtein Score)\n        score = val_score\n\n        if self.best_score is None:\n            self.best_score = score\n            self.save_checkpoint(model)\n        elif score < self.best_score + self.min_delta:\n            self.counter += 1\n            print(f'EarlyStopping counter: {self.counter} out of {self.patience}')\n            if self.counter >= self.patience:\n                self.early_stop = True\n        else:\n            self.best_score = score\n            if self.verbose:\n                print(f'Validation score improved ({self.best_score:.4f}). Saving model...')\n            self.save_checkpoint(model)\n            self.counter = 0\n\n    def save_checkpoint(self, model):\n        '''Saves model when validation score improves.'''\n        # When using DDP, we need to save the underlying model's state dict\n        torch.save(model.module.state_dict(), 'checkpoint_best_score.pth')\n\ndef custom_collate_fn(batch):\n    inputs = [item['input'] for item in batch]\n    token_ids = [item['token_ids'] for item in batch]\n    phrases = [item['phrase'] for item in batch]\n    inputs_padded = torch.nn.utils.rnn.pad_sequence(inputs, batch_first=True, padding_value=0.0)\n    token_ids_stacked = torch.stack(token_ids)\n    return {'input': inputs_padded, 'token_ids': token_ids_stacked, 'phrase': phrases}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-29T09:59:07.70501Z","iopub.execute_input":"2025-07-29T09:59:07.705223Z","iopub.status.idle":"2025-07-29T09:59:07.725811Z","shell.execute_reply.started":"2025-07-29T09:59:07.705198Z","shell.execute_reply":"2025-07-29T09:59:07.72508Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# With Augmentations","metadata":{}},{"cell_type":"code","source":"USE_AUGMENTATIONS = True\nNUM_EPOCHS = 20 # Increased for full dataset\nBATCH_SIZE = 4 # Total batch size, will be split across GPUs\nDATASET_SUBSET_SIZE = 24000 # Set to None to use the full dataset\n\nif __name__ == '__main__':\n    DEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    print(f\"Using device: {DEVICE}\")\n\n    # --- Create CFG Object ---\n    cfg = SimpleNamespace(**{})\n    cfg.symmetry_fp = '/kaggle/input/train-landmarks/symmetry.csv'\n    with open('/kaggle/input/train-landmarks/character_to_prediction_index.json', \"r\") as f: char_to_num = json.load(f)\n    cfg.pad_token, cfg.start_token, cfg.end_token = 'P', 'S', 'E'\n    n=len(char_to_num); char_to_num[cfg.pad_token],char_to_num[cfg.start_token],char_to_num[cfg.end_token]=n,n+1,n+2\n    cfg.tokenizer = [char_to_num, {j:i for i,j in char_to_num.items()}]\n    cfg.max_phrase = 33\n    \n    if USE_AUGMENTATIONS: print(\"--- Running experiment WITH augmentations ---\")\n    else: print(\"--- Running experiment WITHOUT augmentations (Baseline) ---\")\n\n    # --- Data Preparation ---\n    master_df = pd.read_csv('/kaggle/input/asl-final-preprocessed-tensors/final_preprocessed_dataset.csv')\n    if DATASET_SUBSET_SIZE is not None:\n        master_df = master_df.sample(n=DATASET_SUBSET_SIZE, random_state=42).copy()\n    \n    train_split_df, val_split_df = train_test_split(master_df, test_size=0.1, random_state=42, shuffle=True)\n    train_dataset = PreprocessedDataset(train_split_df, cfg, use_augmentations=USE_AUGMENTATIONS)\n    val_dataset = PreprocessedDataset(val_split_df, cfg, use_augmentations=False)\n    train_dataloader = DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True, num_workers=0, pin_memory=True, collate_fn=custom_collate_fn)\n    val_dataloader = DataLoader(val_dataset, batch_size=BATCH_SIZE, shuffle=False, num_workers=0, pin_memory=True, collate_fn=custom_collate_fn)\n    print(f\"Training on {len(train_split_df)} samples, validating on {len(val_split_df)} samples.\")\n\n    # --- Model, Loss, Optimizer ---\n    num_classes = len(cfg.tokenizer[0])\n    model = FullModel(in_channels=390, dim=192, out_dim=num_classes)\n    \n    # ✅ WRAP THE MODEL IN DATAPARALLEL TO USE MULTIPLE GPUS\n    if torch.cuda.device_count() > 1:\n        print(f\"Using {torch.cuda.device_count()} GPUs!\")\n        model = nn.DataParallel(model)\n    \n    model.to(DEVICE)\n    \n    criterion = DetailedCombinedLoss(num_classes=num_classes)\n    optimizer = torch.optim.AdamW(model.parameters(), lr=1e-4)\n    scheduler = torch.optim.lr_scheduler.CosineAnnealingLR(optimizer, T_max=NUM_EPOCHS, eta_min=1e-6)\n\n    # --- History Logging ---\n    history = {\n            'train_loss': [], 'val_loss': [], 'lev_score': [], 'cer': [], 'wer': [],\n            'train_attn_loss': [], 'train_ctc_loss': [], 'learning_rate': []\n        }\n\n    # --- Training Loop ---\n    for epoch in range(NUM_EPOCHS):\n        model.train()\n        total_train_loss = 0\n        total_attn_loss = 0\n        total_ctc_loss = 0 \n        for batch in tqdm(train_dataloader, desc=f\"Epoch {epoch+1}/{NUM_EPOCHS} [Training]\"):\n            input_tensor, targets = batch['input'].to(DEVICE), batch['token_ids'].to(DEVICE)\n            if input_tensor.nelement() == 0: continue\n            attn_targets, loss_targets = targets[:, :-1], targets[:, 1:]\n            optimizer.zero_grad()\n            ctc_logits, attn_logits = model(input_tensor, attn_targets)\n            ctc_input_lengths = torch.full((ctc_logits.size(0),), ctc_logits.size(1), dtype=torch.long)\n            target_lengths = torch.sum(loss_targets != 0, dim=1)\n            loss, loss_attn, loss_ctc = criterion(ctc_logits, attn_logits, loss_targets, ctc_input_lengths, target_lengths)\n            loss.backward()\n            torch.nn.utils.clip_grad_norm_(model.parameters(), 4.0)\n            optimizer.step()\n            total_train_loss += loss.item()\n            total_attn_loss += loss_attn.item() # Accumulate attention loss\n            total_ctc_loss += loss_ctc.item()\n        \n        history['train_loss'].append(total_train_loss / len(train_dataloader))\n        history['learning_rate'].append(optimizer.param_groups[0]['lr'])\n        scheduler.step()\n\n        # --- Validation Loop ---\n        model.eval()\n        total_val_loss, all_preds, all_labels = 0, [], []\n        with torch.no_grad():\n            for batch in tqdm(val_dataloader, desc=f\"Epoch {epoch+1}/{NUM_EPOCHS} [Validation]\"):\n                input_tensor, phrases, targets = batch['input'].to(DEVICE), batch['phrase'], batch['token_ids'].to(DEVICE)\n                if input_tensor.nelement() == 0: continue\n                # Use model.module when wrapped in DataParallel\n                ctc_logits, attn_logits = model.module(input_tensor, targets[:, :-1]) if isinstance(model, nn.DataParallel) else model(input_tensor, targets[:, :-1])\n                val_loss, _, _ = criterion(ctc_logits, attn_logits, targets[:, 1:], torch.full((ctc_logits.size(0),), ctc_logits.size(1)), torch.sum(targets[:, 1:] != 0, dim=1))\n                total_val_loss += val_loss.item()\n                preds = torch.argmax(ctc_logits, dim=-1)\n                for p in preds.cpu():\n                    p_unique = torch.unique_consecutive(p)\n                    p_filtered = p_unique[p_unique != 0]\n                    all_preds.append(\"\".join([cfg.tokenizer[1].get(c.item(), '') for c in p_filtered]))\n                all_labels.extend(phrases)\n        \n            avg_train_loss = total_train_loss / len(train_dataloader)\n            avg_val_loss = total_val_loss / len(val_dataloader)\n            lev_score = get_score(all_labels, all_preds)\n            cer = 1 - lev_score\n            wer = get_wer(all_labels, all_preds)\n\n            history['train_loss'].append(avg_train_loss)\n            history['val_loss'].append(avg_val_loss)\n            history['lev_score'].append(lev_score)\n            history['cer'].append(cer)\n            history['wer'].append(wer)\n            history['train_attn_loss'].append(total_attn_loss / len(train_dataloader))\n            history['train_ctc_loss'].append(total_ctc_loss / len(train_dataloader))\n            history['learning_rate'].append(optimizer.param_groups[0]['lr'])\n\n            print(f\"Epoch {epoch+1}/{NUM_EPOCHS} | Train Loss: {avg_train_loss:.4f} | Val Loss: {avg_val_loss:.4f} | CER: {cer:.4f} | WER: {wer:.4f} | Levenshtein Score: {lev_score:.4f}\")\n            print(f\"    -> Train Attn Loss: {history['train_attn_loss'][-1]:.4f} | Train CTC Loss: {history['train_ctc_loss'][-1]:.4f}\")\n\n\n    # --- Save Results ---\n    run_type = \"with_augs\" if USE_AUGMENTATIONS else \"no_augs\"\n    with open(f'training_history_{run_type}.json', 'w') as f: json.dump(history, f, indent=4)\n    with open(f'final_predictions_{run_type}.json', 'w') as f: json.dump({'labels': all_labels, 'predictions': all_preds}, f, indent=4)\n    print(f\"\\n--- Saved results for {run_type} run ---\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-29T09:59:07.727369Z","iopub.execute_input":"2025-07-29T09:59:07.727698Z","iopub.status.idle":"2025-07-29T10:45:39.056562Z","shell.execute_reply.started":"2025-07-29T09:59:07.727677Z","shell.execute_reply":"2025-07-29T10:45:39.055861Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Visualizations","metadata":{}},{"cell_type":"code","source":"import json\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\nimport pandas as pd\nimport numpy as np\n\n# Set a consistent style for plots\nsns.set_style(\"whitegrid\")\nplt.rcParams.update({'font.size': 12}) # Adjust base font size for better readability\n\n# --- Load all saved results from your two experiments ---\ntry:\n    with open('training_history_with_augs.json', 'r') as f:\n        history_aug = json.load(f)\n        \n    with open('/kaggle/input/graphs/training_history_no_augs.json', 'r') as f:\n        history_no_augs = json.load(f)\n    \n    with open('final_predictions_with_augs.json', 'r') as f:\n        preds_aug = json.load(f)\n        \n    with open('/kaggle/input/graphs/final_predictions_no_augs.json', 'r') as f:\n        preds_no_augs = json.load(f)\n\n    print(\"--- All result files loaded successfully. Generating plots. ---\")\n\n    # Determine the number of epochs (assuming both histories have the same length)\n    epochs = range(1, len(history_aug['train_loss']) + 1)\n\n    # --- Figure 1: Core Performance Metrics (Loss, Levenshtein, CER/WER, Learning Rate) ---\n    fig1, axes1 = plt.subplots(2, 2, figsize=(20, 15)) # 2x2 grid for 4 plots\n    fig1.suptitle('Training Performance Analysis: With vs. Without Augmentations', fontsize=20, y=1.02)\n\n    # Plot 1: Loss Curves (Training and Validation)\n    ax1 = axes1[0, 0]\n    ax1.plot(epochs, history_aug['train_loss'], 'g-', label='Train Loss (With Augs)')\n    ax1.plot(epochs, history_aug['val_loss'], 'g--', label='Val Loss (With Augs)')\n    ax1.plot(epochs, history_no_augs['train_loss'], 'b-', label='Train Loss (Baseline)')\n    ax1.plot(epochs, history_no_augs['val_loss'], 'b--', label='Val Loss (Baseline)')\n    ax1.set_title('Loss vs. Epochs', fontsize=14)\n    ax1.set_xlabel('Epoch')\n    ax1.set_ylabel('Loss')\n    ax1.legend()\n    ax1.grid(True)\n\n    # Plot 2: Levenshtein Score Curves (Validation)\n    ax2 = axes1[0, 1]\n    ax2.plot(epochs, history_aug['lev_score'], 'g-', label='Levenshtein Score (With Augs)')\n    ax2.plot(epochs, history_no_augs['lev_score'], 'b-', label='Levenshtein Score (Baseline)')\n    ax2.set_title('Validation Levenshtein Score vs. Epochs', fontsize=14)\n    ax2.set_xlabel('Epoch')\n    ax2.set_ylabel('Levenshtein Score')\n    ax2.legend()\n    ax2.grid(True)\n    \n    # Plot 3: Character Error Rate (CER) and Word Error Rate (WER)\n    ax3 = axes1[1, 0]\n    ax3.plot(epochs, history_aug['cer'], 'g-', label='CER (With Augs)')\n    ax3.plot(epochs, history_no_augs['cer'], 'b-', label='CER (Baseline)')\n    ax3.plot(epochs, history_aug['wer'], 'g--', label='WER (With Augs)')\n    ax3.plot(epochs, history_no_augs['wer'], 'b--', label='WER (Baseline)')\n    ax3.set_title('CER and WER vs. Epochs', fontsize=14)\n    ax3.set_xlabel('Epoch')\n    ax3.set_ylabel('Error Rate')\n    ax3.legend()\n    ax3.grid(True)\n\n    # Plot 4: Learning Rate Schedule\n    ax4 = axes1[1, 1]\n    # Assuming learning rate schedule is identical for both runs if NUM_EPOCHS is the same\n    ax4.plot(epochs, history_aug.get('learning_rate', []), 'r-', label='Learning Rate')\n    ax4.set_title('Learning Rate vs. Epochs', fontsize=14)\n    ax4.set_xlabel('Epoch')\n    ax4.set_ylabel('Learning Rate')\n    ax4.legend()\n    ax4.grid(True)\n    \n    plt.tight_layout(rect=[0, 0, 1, 0.98]) # Adjust rect to make space for suptitle\n    plt.savefig('performance_metrics_comparison.png', dpi=300, bbox_inches='tight')\n    plt.show()\n    print(\"Saved 'performance_metrics_comparison.png'\")\n\n    # --- Figure 2: Training Component Losses (CTC and Attention Loss) ---\n    fig2, ax_comp = plt.subplots(1, 1, figsize=(10, 7))\n    fig2.suptitle('Training CTC and Attention Loss Components', fontsize=16, y=1.02)\n\n    ax_comp.plot(epochs, history_aug['train_ctc_loss'], 'g-', label='Train CTC Loss (With Augs)')\n    ax_comp.plot(epochs, history_aug['train_attn_loss'], 'g--', label='Train Attn Loss (With Augs)')\n    ax_comp.plot(epochs, history_no_augs['train_ctc_loss'], 'b-', label='Train CTC Loss (Baseline)')\n    ax_comp.plot(epochs, history_no_augs['train_attn_loss'], 'b--', label='Train Attn Loss (Baseline)')\n    ax_comp.set_title('Component Losses vs. Epochs', fontsize=14)\n    ax_comp.set_xlabel('Epoch')\n    ax_comp.set_ylabel('Loss')\n    ax_comp.legend()\n    ax_comp.grid(True)\n\n    plt.tight_layout(rect=[0, 0, 1, 0.98])\n    plt.savefig('component_losses_comparison.png', dpi=300, bbox_inches='tight')\n    plt.show()\n    print(\"Saved 'component_losses_comparison.png'\")\n\n\n    # --- Plot 3: Confusion Matrix for the better performing model ---\n    # Determine which model performed better based on the maximum Levenshtein score\n    best_score_aug = max(history_aug['lev_score'])\n    best_score_no_augs = max(history_no_augs['lev_score'])\n\n    if best_score_aug > best_score_no_augs:\n        print(f\"\\n--- Generating Confusion Matrix for the Augmented Model (Best Levenshtein Score: {best_score_aug:.4f}) ---\")\n        all_labels = preds_aug['labels']\n        all_preds = preds_aug['predictions']\n        model_name = \"Augmented Model\"\n    else:\n        print(f\"\\n--- Generating Confusion Matrix for the Baseline Model (Best Levenshtein Score: {best_score_no_augs:.4f}) ---\")\n        all_labels = preds_no_augs['labels']\n        all_preds = preds_no_augs['predictions']\n        model_name = \"Baseline Model\"\n\n    # Flatten all characters into a single list for the confusion matrix\n    flat_labels = [char for phrase in all_labels for char in phrase]\n    flat_preds = [char for phrase in all_preds for char in phrase]\n    \n    # Trim to the same length to handle any minor prediction length differences\n    min_len = min(len(flat_labels), len(flat_preds))\n    flat_labels, flat_preds = flat_labels[:min_len], flat_preds[:min_len]\n\n    # Get a sorted list of unique characters that appear in the data\n    # This ensures the confusion matrix includes all possible characters and maintains order\n    labels = sorted(list(set(flat_labels) | set(flat_preds)))\n    \n    # Create the confusion matrix\n    cm = confusion_matrix(flat_labels, flat_preds, labels=labels)\n    \n    plt.figure(figsize=(22, 18)) # Adjust size for potentially many characters\n    sns.heatmap(cm, annot=True, fmt='d', xticklabels=labels, yticklabels=labels, cmap='viridis', cbar_kws={'label': 'Count'})\n    plt.title(f'Character-Level Confusion Matrix ({model_name})', fontsize=18)\n    plt.xlabel('Predicted Characters', fontsize=14)\n    plt.ylabel('True Characters', fontsize=14)\n    plt.xticks(rotation=90) # Rotate x-axis labels for better readability if many characters\n    plt.yticks(rotation=0)  # Keep y-axis labels horizontal\n    plt.tight_layout()\n    plt.savefig('confusion_matrix.png', dpi=300, bbox_inches='tight')\n    plt.show()\n    print(\"Saved 'confusion_matrix.png'\")\n\nexcept FileNotFoundError:\n    print(\"Could not find the necessary .json result files. Please ensure 'training_history_with_augs.json', 'training_history_no_augs.json', 'final_predictions_with_augs.json', and 'final_predictions_no_augs.json' are in the same directory.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-29T12:47:55.963082Z","iopub.execute_input":"2025-07-29T12:47:55.963429Z","iopub.status.idle":"2025-07-29T12:47:56.92093Z","shell.execute_reply.started":"2025-07-29T12:47:55.963404Z","shell.execute_reply":"2025-07-29T12:47:56.920061Z"}},"outputs":[],"execution_count":null}]}