{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":112899,"databundleVersionId":13449579,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd \nfrom glob import glob\nimport os \nimport matplotlib.pyplot as plt\nimport torch\nfrom torch.utils.data import Dataset\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\nimport numpy as np\nfrom collections import Counter\nfrom torch.utils.data import DataLoader\nimport time\nfrom sklearn.metrics import roc_auc_score\nfrom datetime import datetime","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:01.3284Z","iopub.execute_input":"2025-09-04T14:43:01.329243Z","iopub.status.idle":"2025-09-04T14:43:01.334075Z","shell.execute_reply.started":"2025-09-04T14:43:01.329211Z","shell.execute_reply":"2025-09-04T14:43:01.333231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"metadata_path=\"/kaggle/input/grand-xray-slam-division-a/train1.csv\"\nimage_folder=\"/kaggle/input/grand-xray-slam-division-a/train1\"\ndf=pd.read_csv(metadata_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:01.345073Z","iopub.execute_input":"2025-09-04T14:43:01.345441Z","iopub.status.idle":"2025-09-04T14:43:01.578371Z","shell.execute_reply.started":"2025-09-04T14:43:01.345423Z","shell.execute_reply":"2025-09-04T14:43:01.577816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:01.579762Z","iopub.execute_input":"2025-09-04T14:43:01.580043Z","iopub.status.idle":"2025-09-04T14:43:01.630454Z","shell.execute_reply.started":"2025-09-04T14:43:01.580016Z","shell.execute_reply":"2025-09-04T14:43:01.629699Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:01.631298Z","iopub.execute_input":"2025-09-04T14:43:01.631837Z","iopub.status.idle":"2025-09-04T14:43:01.657145Z","shell.execute_reply.started":"2025-09-04T14:43:01.631816Z","shell.execute_reply":"2025-09-04T14:43:01.656513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:01.658782Z","iopub.execute_input":"2025-09-04T14:43:01.659051Z","iopub.status.idle":"2025-09-04T14:43:01.694271Z","shell.execute_reply.started":"2025-09-04T14:43:01.659034Z","shell.execute_reply":"2025-09-04T14:43:01.693662Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CONDITIONS = [\n    'Atelectasis', 'Cardiomegaly', 'Consolidation', 'Edema',\n    'Enlarged Cardiomediastinum', 'Fracture', 'Lung Lesion', 'Lung Opacity',\n    'No Finding', 'Pleural Effusion', 'Pleural Other', 'Pneumonia',\n    'Pneumothorax', 'Support Devices'\n]\n\nlabel_counts = df[CONDITIONS].sum().sort_values(ascending=False)\nprint(\"Label distribution:\")\nprint(label_counts)\n\n# Visualize distribution\nplt.figure(figsize=(12, 6))\nlabel_counts.plot(kind='bar')\nplt.title('Distribution of Thoracic Conditions')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:01.6949Z","iopub.execute_input":"2025-09-04T14:43:01.695066Z","iopub.status.idle":"2025-09-04T14:43:01.884768Z","shell.execute_reply.started":"2025-09-04T14:43:01.695053Z","shell.execute_reply":"2025-09-04T14:43:01.884089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ChestXrayDataset(Dataset):\n    def __init__(self, dataframe, image_folder, conditions, transform=None):\n        self.dataframe = dataframe.reset_index(drop=True)  # Clean the indices\n        self.image_folder = image_folder  # Folder where images are stored\n        self.conditions = conditions      # List of condition names\n        self.transform = transform\n        \n        print(f\"Dataset created with {len(self.dataframe)} samples\")\n        \n    def __len__(self):\n        return len(self.dataframe)\n    \n    def __getitem__(self, idx):\n        # Get the row using iloc (this ensures we get the idx-th row regardless of index values)\n        row = self.dataframe.iloc[idx]\n        img_filename = row['Image_name']\n        img_path = os.path.join(self.image_folder, img_filename)\n        if not os.path.exists(img_path):\n            print(f\" File not found: {img_path}\")\n            image = Image.new('RGB', (224, 224), color=0)\n        else:\n            try:\n                image = Image.open(img_path).convert(\"RGB\")\n            except Exception as e:\n                print(f\" Error loading {img_path}: {e}\")\n                image = Image.new('RGB', (224, 224), color=0)\n        label_values = []\n        for condition in self.conditions:\n            label_values.append(float(row[condition]))\n        \n        label = torch.tensor(label_values, dtype=torch.float32)\n        if self.transform:\n            image = self.transform(image)\n        \n        return image, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:01.885496Z","iopub.execute_input":"2025-09-04T14:43:01.885798Z","iopub.status.idle":"2025-09-04T14:43:01.89305Z","shell.execute_reply.started":"2025-09-04T14:43:01.885775Z","shell.execute_reply":"2025-09-04T14:43:01.892337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchvision import transforms\ntrain_transform = transforms.Compose([\n    transforms.Resize((224, 224)),  # Resize images to 224x224\n    transforms.RandomHorizontalFlip(),  # Random horizontal flip\n    transforms.RandomRotation(20),  # Random rotation\n    transforms.ToTensor(),  # Convert to tensor\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])  # Normalize\n])\n\n# Define transforms for validation and testing (no augmentation)\nval_test_transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:01.893844Z","iopub.execute_input":"2025-09-04T14:43:01.894045Z","iopub.status.idle":"2025-09-04T14:43:01.910566Z","shell.execute_reply.started":"2025-09-04T14:43:01.89403Z","shell.execute_reply":"2025-09-04T14:43:01.909958Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dataset Subset Strategy for Efficient Development\n\nDue to the large dataset size impacting development velocity, we'll create a stratified \nsubset that maintains the original distribution of:\n- Label frequencies (preserving class balance for multi-label scenarios)  \n- Data sources (ensuring representation across different hospitals/scanners)\n\nThis subset approach enables:\n- Rapid prototyping and architecture comparison\n- Hyperparameter tuning with faster iteration cycles\n- Technique validation before scaling to full dataset\n\n**Workflow:** Develop on subset → Validate findings → Scale to full dataset","metadata":{}},{"cell_type":"code","source":"def detect_institution_from_filename(image_name):\n    \"\"\"\n    Extract institution from filename pattern using string prefix approach\n    CheXpert: 00000001_001_001.jpg (starts with 00 or other patterns < 10)\n    MIMIC: 10001217_001_001.jpg (starts with 10)  \n    NIH: 20009281_016_000.jpg (starts with 20)\n    \"\"\"\n    first_part = image_name.split('_')[0]\n    \n    if first_part.startswith('10'):\n        return 'MIMIC' \n    elif first_part.startswith('20'):\n        return 'NIH'\n    else:\n        return 'CheXpert'\n\ndef create_stratified_subset(df, CONDITIONS, target_size=10000, random_state=42):\n    \"\"\"\n    Create subset maintaining exact label and institution proportions\n    \"\"\"\n    np.random.seed(random_state)\n    \n    # Create working copy and add institution column from filename\n    df = df.copy()\n    df['Institution'] = df['Image_name'].apply(detect_institution_from_filename)\n    \n    # Calculate sampling ratio\n    total_size = len(df)\n    sampling_ratio = target_size / total_size\n    \n    print(f\"Original dataset size: {total_size}\")\n    print(f\"Target subset size: {target_size}\")\n    print(f\"Sampling ratio: {sampling_ratio:.4f}\")\n    print(\"-\" * 50)\n    \n    # Show institution distribution\n    print(\"Institution Distribution (from filename analysis):\")\n    inst_counts = df['Institution'].value_counts()\n    for inst, count in inst_counts.items():\n        pct = (count / len(df)) * 100\n        print(f\"  {inst}: {count:,} samples ({pct:.1f}%)\")\n    print(\"-\" * 50)\n    \n    # Create label pattern for multi-label stratification\n    df['label_pattern'] = df[CONDITIONS].apply(\n        lambda row: ''.join(row.astype(str)), axis=1\n    )\n    \n    # Group by institution and label pattern\n    grouped = df.groupby(['Institution', 'label_pattern'])\n    \n    print(f\"Total unique strata (Institution + Label Pattern): {len(grouped)}\")\n    \n    subset_parts = []\n    skipped_strata = 0\n    \n    for (institution, pattern), group in grouped:\n        # Calculate target size for this stratum\n        stratum_target = max(1, int(len(group) * sampling_ratio))\n        actual_sample = min(stratum_target, len(group))\n        \n        if actual_sample > 0:\n            sampled = group.sample(n=actual_sample, random_state=random_state)\n            subset_parts.append(sampled)\n        else:\n            skipped_strata += 1\n    \n    print(f\"Strata sampled: {len(subset_parts)}\")\n    print(f\"Strata skipped (too small): {skipped_strata}\")\n    \n    # Combine all parts\n    subset_df = pd.concat(subset_parts).reset_index(drop=True)\n    \n    # Clean up temporary columns\n    subset_df = subset_df.drop(['Institution', 'label_pattern'], axis=1)\n    df = df.drop(['Institution', 'label_pattern'], axis=1)\n    \n    print(f\"Final subset size: {len(subset_df)}\")\n    \n    return subset_df\n\ndef validate_subset_distributions(original_df, subset_df, CONDITIONS):\n    \"\"\"\n    Compare distributions between original and subset datasets\n    \"\"\"\n    print(\"DISTRIBUTION VALIDATION\")\n    print(\"=\" * 60)\n    \n    # Add institution columns for both datasets using filename method\n    for df_name, df in [(\"original\", original_df), (\"subset\", subset_df)]:\n        df['Institution'] = df['Image_name'].apply(detect_institution_from_filename)\n    \n    # 1. Label Distribution Comparison\n    print(\"1. LABEL DISTRIBUTION COMPARISON:\")\n    print(\"-\" * 40)\n    print(f\"{'Condition':<25} {'Original %':<12} {'Subset %':<12} {'Difference':<12}\")\n    print(\"-\" * 40)\n    \n    max_diff = 0\n    for condition in CONDITIONS:\n        orig_pct = original_df[condition].mean() * 100\n        subset_pct = subset_df[condition].mean() * 100\n        diff = abs(orig_pct - subset_pct)\n        max_diff = max(max_diff, diff)\n        \n        print(f\"{condition:<25} {orig_pct:<12.2f} {subset_pct:<12.2f} {diff:<12.2f}\")\n    \n    print(f\"\\nMaximum label difference: {max_diff:.2f}%\")\n    \n    # 2. Institution Distribution Comparison\n    print(f\"\\n2. INSTITUTION DISTRIBUTION COMPARISON:\")\n    print(\"-\" * 40)\n    \n    orig_inst = original_df['Institution'].value_counts(normalize=True) * 100\n    subset_inst = subset_df['Institution'].value_counts(normalize=True) * 100\n    \n    print(f\"{'Institution':<15} {'Original %':<12} {'Subset %':<12} {'Difference':<12}\")\n    print(\"-\" * 40)\n    \n    for institution in ['CheXpert', 'MIMIC', 'NIH']:\n        orig_pct = orig_inst.get(institution, 0)\n        subset_pct = subset_inst.get(institution, 0)\n        diff = abs(orig_pct - subset_pct)\n        print(f\"{institution:<15} {orig_pct:<12.2f} {subset_pct:<12.2f} {diff:<12.2f}\")\n    \n    # 3. Multi-label Pattern Analysis\n    print(f\"\\n3. MULTI-LABEL PATTERN ANALYSIS:\")\n    print(\"-\" * 40)\n    \n    # Count number of positive conditions per image\n    original_df['num_conditions'] = original_df[CONDITIONS].sum(axis=1)\n    subset_df['num_conditions'] = subset_df[CONDITIONS].sum(axis=1)\n    \n    orig_pattern = original_df['num_conditions'].value_counts(normalize=True).sort_index() * 100\n    subset_pattern = subset_df['num_conditions'].value_counts(normalize=True).sort_index() * 100\n    \n    print(f\"{'# Conditions':<15} {'Original %':<12} {'Subset %':<12} {'Difference':<12}\")\n    print(\"-\" * 40)\n    \n    all_counts = set(orig_pattern.index) | set(subset_pattern.index)\n    for count in sorted(all_counts):\n        orig_pct = orig_pattern.get(count, 0)\n        subset_pct = subset_pattern.get(count, 0)\n        diff = abs(orig_pct - subset_pct)\n        print(f\"{count:<15} {orig_pct:<12.2f} {subset_pct:<12.2f} {diff:<12.2f}\")\n    \n    # 4. Summary Statistics\n    print(f\"\\n4. SUMMARY STATISTICS:\")\n    print(\"-\" * 40)\n    print(f\"Original dataset size: {len(original_df):,}\")\n    print(f\"Subset size: {len(subset_df):,}\")\n    print(f\"Sampling ratio: {len(subset_df)/len(original_df):.4f}\")\n    print(f\"Average conditions per image (Original): {original_df[CONDITIONS].sum(axis=1).mean():.2f}\")\n    print(f\"Average conditions per image (Subset): {subset_df[CONDITIONS].sum(axis=1).mean():.2f}\")\n    \n    # Clean up temporary columns\n    for df in [original_df, subset_df]:\n        df.drop(['Institution', 'num_conditions'], axis=1, inplace=True)\n    \n    print(f\"\\n5. QUALITY ASSESSMENT:\")\n    print(\"-\" * 40)\n    if max_diff < 1.0:\n        print(\"EXCELLENT: Label distributions match very closely (<1% difference)\")\n    elif max_diff < 2.0:\n        print(\"GOOD: Label distributions acceptable (<2% difference)\")\n    elif max_diff < 5.0:\n        print(\"CAUTION: Some label distributions differ significantly (2-5% difference)\")\n    else:\n        print(\"WARNING: Large distribution differences (>5%). Consider larger subset or different sampling strategy.\")\n\n# Test the detection function first\nprint(\"Testing institution detection function:\")\ntest_filenames = ['00000001_001_001.jpg', '10001217_001_001.jpg', '20009281_016_000.jpg']\nfor filename in test_filenames:\n    result = detect_institution_from_filename(filename)\n    print(f\"{filename} -> {result}\")\n\n# Test on actual data\ndf['Institution_Test'] = df['Image_name'].apply(detect_institution_from_filename)\nprint(f\"\\nActual institution distribution in your data:\")\nprint(df['Institution_Test'].value_counts())\nprint(f\"Percentages:\")\nprint(df['Institution_Test'].value_counts(normalize=True) * 100)\n\nprint(\"\\n\" + \"=\"*60)\nprint(\"Creating stratified subset with corrected institution detection...\")\ntrain_subset = create_stratified_subset(df, CONDITIONS, target_size=10000, random_state=42)\n\nprint(\"\\nValidating subset quality...\")\nvalidate_subset_distributions(df, train_subset, CONDITIONS)\n\nprint(f\"\\nSubset created successfully!\")\nprint(f\"Use 'train_subset' for your experiments.\")\nprint(f\"Original shape: {df.shape}\")\nprint(f\"Subset shape: {train_subset.shape}\")\n\n# Clean up the test column\ndf = df.drop('Institution_Test', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:01.911355Z","iopub.execute_input":"2025-09-04T14:43:01.911547Z","iopub.status.idle":"2025-09-04T14:43:07.82918Z","shell.execute_reply.started":"2025-09-04T14:43:01.911533Z","shell.execute_reply":"2025-09-04T14:43:07.828454Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df, val_df = train_test_split(df, test_size=0.25, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:07.829872Z","iopub.execute_input":"2025-09-04T14:43:07.830052Z","iopub.status.idle":"2025-09-04T14:43:07.854255Z","shell.execute_reply.started":"2025-09-04T14:43:07.830036Z","shell.execute_reply":"2025-09-04T14:43:07.853669Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = ChestXrayDataset(train_df, image_folder, CONDITIONS, train_transform)\nval_dataset = ChestXrayDataset(val_df,image_folder, CONDITIONS,  transform=val_test_transform)\n#test_dataset = ChestXrayDataset(test_df, transform=val_test_transform)\ntrain_loader = DataLoader(\n    train_dataset, \n    batch_size=64,       \n    shuffle=True,\n    num_workers=4,       \n    pin_memory=True,       \n    persistent_workers=True, \n    prefetch_factor=2    \n)\n\nval_loader = DataLoader(\n    val_dataset, \n    batch_size=64,        \n    shuffle=False,\n    num_workers=4,        \n    pin_memory=True,\n    persistent_workers=True,\n    prefetch_factor=2\n)\n#test_loader = DataLoader(test_dataset, batch_size=32, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:07.856403Z","iopub.execute_input":"2025-09-04T14:43:07.856584Z","iopub.status.idle":"2025-09-04T14:43:08.06405Z","shell.execute_reply.started":"2025-09-04T14:43:07.856571Z","shell.execute_reply":"2025-09-04T14:43:08.063242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_images, test_labels = next(iter(train_loader))\nprint(f\"Batch shape: {test_images.shape}\")\nprint(f\"Labels shape: {test_labels.shape}\")\nprint(f\"Image range: {test_images.min():.3f} to {test_images.max():.3f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:08.065035Z","iopub.execute_input":"2025-09-04T14:43:08.065329Z","iopub.status.idle":"2025-09-04T14:43:14.568245Z","shell.execute_reply.started":"2025-09-04T14:43:08.065304Z","shell.execute_reply":"2025-09-04T14:43:14.566961Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def show_samples(dataloader, conditions, num_samples=8, figsize=(16, 12)):\n    \"\"\"\n    Display sample images with their labels from the dataloader\n    \"\"\"\n    # Get one batch\n    images, labels = next(iter(dataloader))\n    \n    # Take only the requested number of samples\n    images = images[:num_samples]\n    labels = labels[:num_samples]\n    \n    # Calculate grid dimensions\n    cols = 4\n    rows = (num_samples + cols - 1) // cols\n    \n    fig, axes = plt.subplots(rows, cols, figsize=figsize)\n    if rows == 1:\n        axes = [axes]\n    if cols == 1:\n        axes = [[ax] for ax in axes]\n    \n    for idx in range(num_samples):\n        row = idx // cols\n        col = idx % cols\n        ax = axes[row][col]\n        \n        # Denormalize the image for display\n        # Reverse ImageNet normalization\n        image = images[idx]\n        mean = torch.tensor([0.485, 0.456, 0.406]).view(3, 1, 1)\n        std = torch.tensor([0.229, 0.224, 0.225]).view(3, 1, 1)\n        \n        # Denormalize\n        image = image * std + mean\n        image = torch.clamp(image, 0, 1)\n        \n        # Convert to numpy and transpose for matplotlib (H, W, C)\n        image_np = image.permute(1, 2, 0).numpy()\n        \n        # Display image\n        ax.imshow(image_np)\n        ax.axis('off')\n        \n        # Get positive conditions for this sample\n        sample_labels = labels[idx]\n        positive_conditions = [conditions[i] for i, val in enumerate(sample_labels) if val > 0.5]\n        \n        # Create title with positive conditions\n        if positive_conditions:\n            title = ', '.join(positive_conditions[:3]) \n            if len(positive_conditions) > 3:\n                title += f' (+{len(positive_conditions)-3} more)'\n        else:\n            title = 'No positive findings'\n        \n        ax.set_title(title, fontsize=10, pad=5)\n    \n\n    for idx in range(num_samples, rows * cols):\n        row = idx // cols\n        col = idx % cols\n        axes[row][col].axis('off')\n    \n    plt.tight_layout()\n    plt.show()\n\n\nshow_samples(train_loader, CONDITIONS, num_samples=8)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:14.570336Z","iopub.execute_input":"2025-09-04T14:43:14.570737Z","iopub.status.idle":"2025-09-04T14:43:31.551455Z","shell.execute_reply.started":"2025-09-04T14:43:14.570635Z","shell.execute_reply":"2025-09-04T14:43:31.550756Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Densenet 121 ","metadata":{}},{"cell_type":"code","source":"\"\"\"\n# Load pre-trained DenseNet121\nmodel = models.densenet121(pretrained=True)\n\n# Modify the classifier layer for multi-label classification\nnum_features = model.classifier.in_features\nmodel.classifier = nn.Linear(num_features, len(CONDITIONS))\n\n# Use DataParallel if multiple GPUs are available\nif torch.cuda.device_count() > 1:\n    print(f\"Using {torch.cuda.device_count()} GPUs!\")\n    model = nn.DataParallel(model)\n\n# Move the model to the GPU(s)\nmodel = model.cuda()\n\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:31.552096Z","iopub.execute_input":"2025-09-04T14:43:31.55235Z","iopub.status.idle":"2025-09-04T14:43:31.56427Z","shell.execute_reply.started":"2025-09-04T14:43:31.55233Z","shell.execute_reply":"2025-09-04T14:43:31.563369Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Efficient Net B4","metadata":{}},{"cell_type":"code","source":"\"\"\"\nfrom efficientnet_pytorch import EfficientNet \nmodel = EfficientNet.from_pretrained('efficientnet-b4')\n\n# Get number of features in the last layer\nnum_features = model._fc.in_features\n\n# Replace the classifier with your own\nmodel._fc = nn.Linear(num_features, len(CONDITIONS))\n\n# Use DataParallel if multiple GPUs are available\nif torch.cuda.device_count() > 1:\n    print(f\"Using {torch.cuda.device_count()} GPUs!\")\n    model = nn.DataParallel(model)\n\n# Move model to GPU(s)\nmodel = model.cuda()\n\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:31.565103Z","iopub.execute_input":"2025-09-04T14:43:31.566909Z","iopub.status.idle":"2025-09-04T14:43:31.582412Z","shell.execute_reply.started":"2025-09-04T14:43:31.566885Z","shell.execute_reply":"2025-09-04T14:43:31.581632Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Resnet 50","metadata":{}},{"cell_type":"code","source":"\nmodel_resnet = models.resnet50(pretrained=True)\nnum_features = model_resnet.fc.in_features\nmodel_resnet.fc = nn.Sequential(\n    nn.Dropout(p=0.3), \n    nn.Linear(num_features, len(CONDITIONS))\n)\n# Use DataParallel if multiple GPUs are available\nif torch.cuda.device_count() > 1:\n    print(f\"Using {torch.cuda.device_count()} GPUs for ResNet-50!\")\n    model_resnet = nn.DataParallel(model_resnet)\n# Move the model to the GPU(s)\nmodel_resnet = model_resnet.cuda()\nmodel=model_resnet\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:31.583107Z","iopub.execute_input":"2025-09-04T14:43:31.583387Z","iopub.status.idle":"2025-09-04T14:43:32.43696Z","shell.execute_reply.started":"2025-09-04T14:43:31.583364Z","shell.execute_reply":"2025-09-04T14:43:32.435761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\ndef calculate_class_weights_from_dataloader(train_loader, num_classes):\n\n    class_counts = torch.zeros(num_classes)\n    total_samples = 0\n    \n    for _, labels in train_loader:\n        # labels shape: [batch_size, num_classes]\n        class_counts += labels.sum(dim=0)\n        total_samples += labels.shape[0]\n    \n    # Calculate frequencies\n    class_frequencies = class_counts / total_samples\n    \n    # Calculate pos_weights (higher weight for rare classes)\n    pos_weights = (1.0 - class_frequencies) / (class_frequencies + 1e-8)\n    \n    print(\"Class frequencies:\", class_frequencies)\n    print(\"Pos weights:\", pos_weights)\n    \n    return pos_weights.cuda()\n\n# Use with your data\nnum_classes = len(CONDITIONS)  # Based on your dataset\npos_weights = calculate_class_weights_from_dataloader(train_loader, num_classes)\n\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:32.437756Z","iopub.execute_input":"2025-09-04T14:43:32.439897Z","iopub.status.idle":"2025-09-04T14:43:32.448527Z","shell.execute_reply.started":"2025-09-04T14:43:32.439872Z","shell.execute_reply":"2025-09-04T14:43:32.447779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch.optim as optim\nimport torch.nn as nn\n\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = optim.Adam(model.parameters(), lr=0.0001)  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:32.450347Z","iopub.execute_input":"2025-09-04T14:43:32.450833Z","iopub.status.idle":"2025-09-04T14:43:32.474106Z","shell.execute_reply.started":"2025-09-04T14:43:32.450809Z","shell.execute_reply":"2025-09-04T14:43:32.470843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_model(model, train_loader, val_loader, criterion, optimizer, num_epochs=10, device='cuda'):\n    \"\"\"\n    Enhanced training function with detailed logging\n    \"\"\"\n    # Track training history\n    train_losses = []\n    val_losses = []\n    val_aucs = []\n    best_val_auc = 0.0\n    \n    print(f\"Starting training for {num_epochs} epochs...\")\n    print(f\"Training batches: {len(train_loader)}\")\n    print(f\"Validation batches: {len(val_loader)}\")\n    print(\"-\" * 70)\n    \n    for epoch in range(num_epochs):\n        epoch_start_time = time.time()\n        \n        # Training phase\n        model.train()\n        train_loss = 0.0\n        train_batches = 0\n        \n        for batch_idx, (images, labels) in enumerate(train_loader):\n            images, labels = images.to(device), labels.to(device)\n            \n            optimizer.zero_grad()\n            outputs = model(images)\n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n            \n            train_loss += loss.item()\n            train_batches += 1\n            \n            # Print progress every 100 batches\n            if (batch_idx + 1) % 100 == 0:\n                current_loss = train_loss / train_batches\n                print(f\"  Batch {batch_idx + 1}/{len(train_loader)} - Loss: {current_loss:.4f}\")\n        \n        avg_train_loss = train_loss / len(train_loader)\n        train_losses.append(avg_train_loss)\n        \n        # Validation phase\n        model.eval()\n        val_loss = 0.0\n        all_labels = []\n        all_predictions = []\n        \n        with torch.no_grad():\n            for images, labels in val_loader:\n                images, labels = images.to(device), labels.to(device)\n                outputs = model(images)\n                loss = criterion(outputs, labels)\n                val_loss += loss.item()\n                \n                # Store for AUC calculation\n                predictions = torch.sigmoid(outputs).cpu().numpy()\n                all_predictions.append(predictions)\n                all_labels.append(labels.cpu().numpy())\n        \n        avg_val_loss = val_loss / len(val_loader)\n        val_losses.append(avg_val_loss)\n        \n        # Calculate validation AUC\n        all_predictions = np.vstack(all_predictions)\n        all_labels = np.vstack(all_labels)\n        \n        try:\n            val_auc_scores = []\n            for i in range(len(CONDITIONS)):\n                if len(np.unique(all_labels[:, i])) > 1:  \n                    auc = roc_auc_score(all_labels[:, i], all_predictions[:, i])\n                    val_auc_scores.append(auc)\n                else:\n                    val_auc_scores.append(0.5)  \n            \n            mean_val_auc = np.mean(val_auc_scores)\n            val_aucs.append(mean_val_auc)\n            \n        except Exception as e:\n            print(f\"Warning: Could not calculate AUC - {e}\")\n            mean_val_auc = 0.0\n            val_aucs.append(0.0)\n        \n        epoch_time = time.time() - epoch_start_time\n        \n        # Print epoch summary\n        print(f\"Epoch {epoch+1}/{num_epochs}\")\n        print(f\"  Train Loss: {avg_train_loss:.4f}\")\n        print(f\"  Val Loss: {avg_val_loss:.4f}\")\n        print(f\"  Val AUC: {mean_val_auc:.4f}\")\n        print(f\"  Time: {epoch_time:.1f}s\")\n        \n        # Save best model\n        if mean_val_auc > best_val_auc:\n            best_val_auc = mean_val_auc\n            torch.save(model.state_dict(), 'best_model.pth')\n            print(f\"  New best model saved! AUC: {best_val_auc:.4f}\")\n        \n        print(\"-\" * 70)\n    \n    print(f\"Training completed!\")\n    print(f\"Best validation AUC: {best_val_auc:.4f}\")\n    \n    return {\n        'train_losses': train_losses,\n        'val_losses': val_losses,\n        'val_aucs': val_aucs,\n        'best_val_auc': best_val_auc\n    }\n\ndef evaluate_model(model, test_loader, CONDITIONS, device='cuda'):\n    \"\"\"\n    Enhanced evaluation function with detailed metrics\n    \"\"\"\n    print(\"Starting model evaluation...\")\n    print(f\"Test batches: {len(test_loader)}\")\n    \n    model.eval()\n    all_labels = []\n    all_predictions = []\n    \n    eval_start_time = time.time()\n    \n    with torch.no_grad():\n        for batch_idx, (images, labels) in enumerate(test_loader):\n            images, labels = images.to(device), labels.to(device)\n            outputs = model(images)\n            \n            # Convert logits to probabilities\n            predictions = torch.sigmoid(outputs).cpu().numpy()\n            all_predictions.append(predictions)\n            all_labels.append(labels.cpu().numpy())\n            \n            if (batch_idx + 1) % 50 == 0:\n                print(f\"  Processed {batch_idx + 1}/{len(test_loader)} batches\")\n    \n    eval_time = time.time() - eval_start_time\n    \n    # Combine all predictions and labels\n    all_predictions = np.vstack(all_predictions)\n    all_labels = np.vstack(all_labels)\n    \n    print(f\"Evaluation completed in {eval_time:.1f}s\")\n    print(f\"Total samples evaluated: {len(all_predictions)}\")\n    print(\"-\" * 70)\n    \n    # Calculate AUC for each condition\n    auc_scores = []\n    print(\"Per-condition AUC scores:\")\n    \n    for i, condition in enumerate(CONDITIONS):\n        try:\n            if len(np.unique(all_labels[:, i])) > 1:\n                auc = roc_auc_score(all_labels[:, i], all_predictions[:, i])\n            else:\n                auc = 0.5\n                print(f\"  Warning: {condition} has only one class in test set\")\n            \n            auc_scores.append(auc)\n            print(f\"  {condition:<25}: {auc:.4f}\")\n            \n        except Exception as e:\n            print(f\"  Error calculating AUC for {condition}: {e}\")\n            auc_scores.append(0.5)\n    \n    mean_auc = np.mean(auc_scores)\n    print(\"-\" * 70)\n    print(f\"Mean AUC: {mean_auc:.4f}\")\n    \n    # Additional statistics\n    print(f\"\\nLabel distribution in test set:\")\n    for i, condition in enumerate(CONDITIONS):\n        pos_count = np.sum(all_labels[:, i])\n        pos_pct = (pos_count / len(all_labels)) * 100\n        print(f\"  {condition:<25}: {pos_count:>5} ({pos_pct:>4.1f}%)\")\n    \n    return {\n        'auc_scores': auc_scores,\n        'mean_auc': mean_auc,\n        'predictions': all_predictions,\n        'labels': all_labels\n    }\n\ndef plot_training_history(history):\n    \"\"\"\n    Plot training curves\n    \"\"\"\n    fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 4))\n    \n    # Loss curves\n    ax1.plot(history['train_losses'], label='Training Loss')\n    ax1.plot(history['val_losses'], label='Validation Loss')\n    ax1.set_title('Training and Validation Loss')\n    ax1.set_xlabel('Epoch')\n    ax1.set_ylabel('Loss')\n    ax1.legend()\n    ax1.grid(True)\n    \n    # AUC curve\n    ax2.plot(history['val_aucs'], label='Validation AUC', color='green')\n    ax2.set_title('Validation AUC')\n    ax2.set_xlabel('Epoch')\n    ax2.set_ylabel('AUC')\n    ax2.legend()\n    ax2.grid(True)\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:32.475199Z","iopub.execute_input":"2025-09-04T14:43:32.478881Z","iopub.status.idle":"2025-09-04T14:43:32.520086Z","shell.execute_reply.started":"2025-09-04T14:43:32.478857Z","shell.execute_reply":"2025-09-04T14:43:32.51937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Training\nhistory = train_model(model, train_loader, val_loader, criterion, optimizer, num_epochs=3)\n\n# Plot training curves\nplot_training_history(history)\n\n# Evaluation\n#results = evaluate_model(model, test_loader, CONDITIONS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T14:43:32.520657Z","iopub.execute_input":"2025-09-04T14:43:32.520941Z","iopub.status.idle":"2025-09-04T16:55:43.601619Z","shell.execute_reply.started":"2025-09-04T14:43:32.520914Z","shell.execute_reply":"2025-09-04T16:55:43.600535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_test_dataset_from_folder(test_folder, CONDITIONS, transform):\n    \"\"\"\n    Create test dataset directly from test folder images\n    \"\"\"\n    # Get all image files from test folder\n    image_files = []\n    for file in os.listdir(test_folder):\n        if file.lower().endswith(('.jpg', '.jpeg', '.png')):\n            image_files.append(file)\n    \n    # Sort to ensure consistent ordering\n    image_files.sort()\n    \n    # Create DataFrame with image names\n    test_df = pd.DataFrame({'Image_name': image_files})\n    \n    # Add dummy labels (required for dataset compatibility)\n    for condition in CONDITIONS:\n        test_df[condition] = 0\n    \n    print(f\"Found {len(test_df)} test images\")\n    \n    # Create dataset\n    test_dataset = ChestXrayDataset(test_df, test_folder, CONDITIONS, transform)\n    \n    return test_dataset, test_df\n\ndef generate_test_predictions(model, test_loader, device='cuda'):\n    \"\"\"\n    Generate predictions for all test images\n    \"\"\"\n    print(\"Generating predictions...\")\n    \n    model.eval()\n    all_predictions = []\n    \n    with torch.no_grad():\n        for batch_idx, (images, *_) in enumerate(test_loader): \n            images = images.to(device)\n            outputs = model(images)\n            \n            # Convert logits to probabilities using sigmoid\n            predictions = torch.sigmoid(outputs).cpu().numpy()\n            all_predictions.append(predictions)\n            \n            if (batch_idx + 1) % 20 == 0:\n                print(f\"  Processed {batch_idx + 1}/{len(test_loader)} batches\")\n    \n    # Combine all predictions\n    all_predictions = np.vstack(all_predictions)\n    print(f\"Generated predictions for {len(all_predictions)} images\")\n    \n    return all_predictions\n\ndef create_submission_csv(test_df, predictions, CONDITIONS, output_path, use_binary=False, threshold=0.5):\n    \"\"\"\n    Create final submission CSV file\n    \"\"\"\n    print(\"Creating submission CSV...\")\n    \n    submission = pd.DataFrame()\n    submission['Image_name'] = test_df['Image_name'].values\n    \n    # Add prediction columns\n    for i, condition in enumerate(CONDITIONS):\n        if use_binary:\n            # Convert probabilities to binary (0 or 1)\n            submission[condition] = (predictions[:, i] > threshold).astype(int)\n        else:\n            # Keep as probabilities (competition requirement)\n            submission[condition] = predictions[:, i]\n    \n    # Show sample output for verification\n    print(f\"Sample predictions for first image:\")\n    print(f\"  Image: {submission.iloc[0]['Image_name']}\")\n    for condition in CONDITIONS:\n        print(f\"  {condition}: {submission.iloc[0][condition]:.4f}\")\n    \n    submission.to_csv(output_path, index=False)\n    print(f\"Submission saved to: {output_path}\")\n    \n    return submission\n\ndef generate_submission(model, test_folder, CONDITIONS, val_test_transform, output_path, device='cuda'):\n    \"\"\"\n    Complete pipeline: test folder -> predictions -> CSV submission\n    \"\"\"\n    print(\"Starting submission generation...\")\n    print(\"=\" * 50)\n    \n    test_dataset, test_df = create_test_dataset_from_folder(test_folder, CONDITIONS, val_test_transform)\n    test_loader = DataLoader(\n        test_dataset,\n        batch_size=128,\n        shuffle=False,  \n        num_workers=4,\n        pin_memory=True,\n        persistent_workers=True,\n        prefetch_factor=2\n    )\n    \n    # Step 3: Generate predictions\n    predictions = generate_test_predictions(model, test_loader, device)\n    \n    # Step 4: Create submission CSV (probabilities by default)\n    submission = create_submission_csv(test_df, predictions, CONDITIONS, output_path)\n    \n    print(\"=\" * 50)\n    print(\"Submission generation completed!\")\n    print(f\"Final submission shape: {submission.shape}\")\n    print(f\"Columns: {list(submission.columns)}\")\n    \n    return submission\n\n# Usage example:\ndef run_final_submission(model, CONDITIONS, val_test_transform):\n    \"\"\"\n    Generate final competition submission\n    \"\"\"\n    # Load best model\n    model.load_state_dict(torch.load('best_model.pth'))\n    model.eval()\n    \n    # Define paths\n    test_folder = \"/kaggle/input/grand-xray-slam-division-a/test1\"\n    \n    # Create timestamped output file\n    timestamp = datetime.now().strftime(\"%Y%m%d_%H%M%S\")\n    output_path = f'/kaggle/working/submission_{timestamp}.csv'\n    \n    # Generate submission with probabilities (correct format)\n    submission = generate_submission(\n        model=model,\n        test_folder=test_folder,\n        CONDITIONS=CONDITIONS,\n        val_test_transform=val_test_transform,\n        output_path=output_path,\n        device='cuda'\n    )\n    \n    print(f\"\\n FINAL SUBMISSION: {output_path}\")\n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T16:55:56.46631Z","iopub.execute_input":"2025-09-04T16:55:56.466615Z","iopub.status.idle":"2025-09-04T16:55:56.478936Z","shell.execute_reply.started":"2025-09-04T16:55:56.466592Z","shell.execute_reply":"2025-09-04T16:55:56.478063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This will create your final competition submission\nfinal_submission = run_final_submission(model, CONDITIONS, val_test_transform)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-04T16:56:03.831165Z","iopub.execute_input":"2025-09-04T16:56:03.831727Z","iopub.status.idle":"2025-09-04T17:14:24.206477Z","shell.execute_reply.started":"2025-09-04T16:56:03.831691Z","shell.execute_reply":"2025-09-04T17:14:24.205788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}