{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":15707590,"datasetId":10062644,"databundleVersionId":16647316}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Chest X-ray Abnormalities - Train Split Inference\n\nThis notebook loads trained Faster R-CNN weights and runs inference on the **training split** obtained from `train_test_split(test_size=0.33)`.","metadata":{}},{"cell_type":"code","source":"import os\nimport warnings\nimport numpy as np\nimport pandas as pd\nfrom skimage import exposure\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, roc_auc_score\n\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\nimport albumentations as A\nfrom albumentations.pytorch.transforms import ToTensorV2\n\nimport torch\nimport torchvision\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision.models.detection.faster_rcnn import FastRCNNPredictor\n\nwarnings.filterwarnings('ignore')\n\nDIR_INPUT = '/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection'\nDIR_TRAIN = f'{DIR_INPUT}/train'\n\n# Update this to your Kaggle dataset that contains model_state.pth\nDIR_WEIGHTS = '/kaggle/input/datasets/urvalatif/cxr-final-model'\nWEIGHTS_FILE = f'{DIR_WEIGHTS}/model_state.pth'\n\n# FDA pipeline-aligned class IDs and thresholds (after labels - 1)\nPTX_CLASS_ID = 12\nEFFUSION_CLASS_ID = 10\nPTX_THRESHOLD = 0.2\nEFFUSION_THRESHOLD = 0.2\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T09:35:32.69424Z","iopub.execute_input":"2026-04-13T09:35:32.694486Z","iopub.status.idle":"2026-04-13T09:35:53.70567Z","shell.execute_reply.started":"2026-04-13T09:35:32.694445Z","shell.execute_reply":"2026-04-13T09:35:53.70508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_annot_df = pd.read_csv(f'{DIR_INPUT}/train.csv')\nall_image_ids = train_annot_df['image_id'].drop_duplicates().values\n\n# Pre-split sanity check over the full dataset (for AUC feasibility)\nfull_labels = train_annot_df[train_annot_df['class_id'] != 14].groupby('image_id')['class_id'].apply(set).reset_index()\nfull_labels['gt_effusion'] = full_labels['class_id'].apply(lambda s: int(11 in s))\nfull_labels['gt_pneumothorax'] = full_labels['class_id'].apply(lambda s: int(13 in s))\n\ndef print_distribution(df, label_col, label_name, scope_name):\n    pos = int(df[label_col].sum())\n    total = int(len(df))\n    neg = int(total - pos)\n    pos_pct = (100.0 * pos / total) if total > 0 else 0.0\n    neg_pct = (100.0 * neg / total) if total > 0 else 0.0\n    print(f'[{scope_name}] {label_name}: pos={pos} ({pos_pct:.2f}%), neg={neg} ({neg_pct:.2f}%), total={total}')\n    if pos == 0 or neg == 0:\n        print(f'WARNING: [{scope_name}] {label_name} has one class only; AUC may be NaN.')\n\nprint_distribution(full_labels, 'gt_effusion', 'Pleural Effusion', 'Full dataset (pre-split)')\nprint_distribution(full_labels, 'gt_pneumothorax', 'Pneumothorax', 'Full dataset (pre-split)')\n\ntrain_image_ids, test_image_ids = train_test_split(\n    all_image_ids,\n    test_size=0.33,\n    random_state=42,\n    shuffle=True\n)\n\ntrain_split_df = pd.DataFrame({'image_id': train_image_ids})\nprint('Total unique train images:', len(all_image_ids))\nprint('Train split images:', len(train_image_ids))\nprint('Test split images:', len(test_image_ids))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T09:35:53.707272Z","iopub.execute_input":"2026-04-13T09:35:53.70765Z","iopub.status.idle":"2026-04-13T09:35:53.931065Z","shell.execute_reply.started":"2026-04-13T09:35:53.707624Z","shell.execute_reply":"2026-04-13T09:35:53.930341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class VinBigInferenceDataset(Dataset):\n    def __init__(self, dataframe, image_dir, transforms=None):\n        super().__init__()\n        self.image_ids = dataframe['image_id'].unique()\n        self.image_dir = image_dir\n        self.transforms = transforms\n\n    def __getitem__(self, index):\n        image_id = self.image_ids[index]\n        dcm_data = pydicom.dcmread(f'{self.image_dir}/{image_id}.dicom')\n        image = apply_voi_lut(dcm_data.pixel_array, dcm_data)\n\n        if dcm_data.PhotometricInterpretation == 'MONOCHROME1':\n            image = np.amax(image) - image\n\n        image = np.stack([image, image, image])\n        image = image - np.min(image)\n        if image.max() > 0:\n            image = image / image.max()\n        image = exposure.equalize_hist(image)\n        image = image.astype('float32')\n        image = image.transpose(1, 2, 0)\n\n        if self.transforms:\n            sample = self.transforms(image=image)\n            image = sample['image']\n\n        return image, image_id\n\n    def __len__(self):\n        return self.image_ids.shape[0]\n\n\ndef get_test_transform():\n    return A.Compose([ToTensorV2()])\n\n\ndef collate_fn(batch):\n    return tuple(zip(*batch))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T09:48:47.242713Z","iopub.execute_input":"2026-04-13T09:48:47.24354Z","iopub.status.idle":"2026-04-13T09:48:47.251844Z","shell.execute_reply.started":"2026-04-13T09:48:47.243494Z","shell.execute_reply":"2026-04-13T09:48:47.251313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device('cuda') if torch.cuda.is_available() else torch.device('cpu')\nnum_classes = 15\n\nmodel = torchvision.models.detection.fasterrcnn_resnet50_fpn(\n    pretrained=False,\n    pretrained_backbone=False,\n)\n\nin_features = model.roi_heads.box_predictor.cls_score.in_features\nmodel.roi_heads.box_predictor = FastRCNNPredictor(in_features, num_classes)\n\nmodel.load_state_dict(torch.load(WEIGHTS_FILE, map_location=device))\nmodel.to(device)\nmodel.eval()\n\ntrain_infer_dataset = VinBigInferenceDataset(train_split_df, DIR_TRAIN, get_test_transform())\ntrain_infer_loader = DataLoader(\n    train_infer_dataset,\n    batch_size=6,\n    shuffle=False,\n    num_workers=2,\n    drop_last=False,\n    collate_fn=collate_fn\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T09:48:49.479564Z","iopub.execute_input":"2026-04-13T09:48:49.480496Z","iopub.status.idle":"2026-04-13T09:48:50.342011Z","shell.execute_reply.started":"2026-04-13T09:48:49.48046Z","shell.execute_reply":"2026-04-13T09:48:50.341188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchvision.models.detection.faster_rcnn import FastRCNNPredictor\nfrom tqdm.auto import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T10:00:45.078297Z","iopub.execute_input":"2026-04-13T10:00:45.07897Z","iopub.status.idle":"2026-04-13T10:00:45.082784Z","shell.execute_reply.started":"2026-04-13T10:00:45.078926Z","shell.execute_reply":"2026-04-13T10:00:45.082082Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# results = []\n\n# with torch.no_grad():\n#     for images, image_ids in train_infer_loader:\n#         images = [image.to(device) for image in images]\n#         outputs = model(images)\n\n#         for i, _ in enumerate(images):\n#             image_id = image_ids[i]\n\n#             labels = outputs[i]['labels'].data.cpu().numpy() - 1\n#             scores = outputs[i]['scores'].data.cpu().numpy()\n\n#             # Keep raw class confidence for AUC; apply threshold only for binary decisions.\n#             ptx_conf = max(\n#                 [scores[j] for j in range(len(labels)) if labels[j] == PTX_CLASS_ID],\n#                 default=0.0\n#             )\n#             eff_conf = max(\n#                 [scores[j] for j in range(len(labels)) if labels[j] == EFFUSION_CLASS_ID],\n#                 default=0.0\n#             )\n\n#             has_pneumothorax = ptx_conf >= PTX_THRESHOLD\n#             has_effusion = eff_conf >= EFFUSION_THRESHOLD\n\n#             if has_pneumothorax and has_effusion:\n#                 classification = 'Pneumothorax & Effusion detected'\n#             elif has_pneumothorax:\n#                 classification = 'Pneumothorax detected'\n#             elif has_effusion:\n#                 classification = 'Effusion detected'\n#             else:\n#                 classification = 'No Finding'\n\n#             results.append({\n#                 'image_id': image_id,\n#                 'classification': classification,\n#                 'has_pneumothorax': bool(has_pneumothorax),\n#                 'has_effusion': bool(has_effusion),\n#                 'ptx_confidence': float(ptx_conf),\n#                 'effusion_confidence': float(eff_conf)\n#             })\n\n# pe_ptx_results_df = pd.DataFrame(results)\n# pe_ptx_results_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T09:48:53.731341Z","iopub.execute_input":"2026-04-13T09:48:53.73205Z","iopub.status.idle":"2026-04-13T09:58:51.029237Z","shell.execute_reply.started":"2026-04-13T09:48:53.732018Z","shell.execute_reply":"2026-04-13T09:58:51.02749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = []\nprocessed_images = 0\n\nwith torch.no_grad():\n    progress = tqdm(train_infer_loader, desc='Inference on train split', unit='batch')\n    for images, image_ids in progress:\n        images = [image.to(device) for image in images]\n        outputs = model(images)\n\n        for i, _ in enumerate(images):\n            image_id = image_ids[i]\n\n            labels = outputs[i]['labels'].data.cpu().numpy() - 1\n            scores = outputs[i]['scores'].data.cpu().numpy()\n\n            # Keep raw class confidence for AUC; apply threshold only for binary decisions.\n            ptx_conf = max(\n                [scores[j] for j in range(len(labels)) if labels[j] == PTX_CLASS_ID],\n                default=0.0\n            )\n            eff_conf = max(\n                [scores[j] for j in range(len(labels)) if labels[j] == EFFUSION_CLASS_ID],\n                default=0.0\n            )\n\n            has_pneumothorax = ptx_conf >= PTX_THRESHOLD\n            has_effusion = eff_conf >= EFFUSION_THRESHOLD\n\n            if has_pneumothorax and has_effusion:\n                classification = 'Pneumothorax & Effusion detected'\n            elif has_pneumothorax:\n                classification = 'Pneumothorax detected'\n            elif has_effusion:\n                classification = 'Effusion detected'\n            else:\n                classification = 'No Finding'\n\n            results.append({\n                'image_id': image_id,\n                'classification': classification,\n                'has_pneumothorax': bool(has_pneumothorax),\n                'has_effusion': bool(has_effusion),\n                'ptx_confidence': float(ptx_conf),\n                'effusion_confidence': float(eff_conf)\n            })\n\n        processed_images += len(image_ids)\n        progress.set_postfix({'processed_images': processed_images})\n\npe_ptx_results_df = pd.DataFrame(results)\npe_ptx_results_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T10:01:27.779995Z","iopub.execute_input":"2026-04-13T10:01:27.780314Z","iopub.status.idle":"2026-04-13T10:02:45.569667Z","shell.execute_reply.started":"2026-04-13T10:01:27.780284Z","shell.execute_reply":"2026-04-13T10:02:45.567969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pe_ptx_results_df.to_csv('pe_ptx_train_split_results.csv', index=False)\nprint('Saved: pe_ptx_train_split_results.csv')\nprint('Rows:', len(pe_ptx_results_df))\nprint('PTX threshold:', PTX_THRESHOLD)\nprint('Effusion threshold:', EFFUSION_THRESHOLD)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T09:58:51.030635Z","iopub.status.idle":"2026-04-13T09:58:51.031012Z","shell.execute_reply.started":"2026-04-13T09:58:51.030819Z","shell.execute_reply":"2026-04-13T09:58:51.030838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Build per-image ground truth for PE and PTX on the same training split\n\n# Original train.csv class ids are 1..14 where 11=Pleural Effusion and 13=Pneumothorax\nEFFUSION_CLASS_ID_TRAINCSV = EFFUSION_CLASS_ID + 1\nPTX_CLASS_ID_TRAINCSV = PTX_CLASS_ID + 1\n\ntrain_labels_only = train_annot_df[train_annot_df['class_id'] != 14].copy()\n\nper_image_gt = train_labels_only.groupby('image_id')['class_id'].apply(set).reset_index()\nper_image_gt['gt_effusion'] = per_image_gt['class_id'].apply(lambda s: int(EFFUSION_CLASS_ID_TRAINCSV in s))\nper_image_gt['gt_pneumothorax'] = per_image_gt['class_id'].apply(lambda s: int(PTX_CLASS_ID_TRAINCSV in s))\nper_image_gt = per_image_gt[['image_id', 'gt_effusion', 'gt_pneumothorax']]\n\ntrain_split_gt = train_split_df.merge(per_image_gt, on='image_id', how='left').fillna(0)\ntrain_split_gt[['gt_effusion', 'gt_pneumothorax']] = train_split_gt[['gt_effusion', 'gt_pneumothorax']].astype(int)\n\n# Post-split sanity check on the train split used for inference\nprint_distribution(train_split_gt, 'gt_effusion', 'Pleural Effusion', 'Train split (inference)')\nprint_distribution(train_split_gt, 'gt_pneumothorax', 'Pneumothorax', 'Train split (inference)')\n\neval_df = pe_ptx_results_df.merge(train_split_gt, on='image_id', how='inner')\n\nprint('Evaluation rows:', len(eval_df))\neval_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T09:35:57.908149Z","iopub.status.idle":"2026-04-13T09:35:57.908402Z","shell.execute_reply.started":"2026-04-13T09:35:57.908279Z","shell.execute_reply":"2026-04-13T09:35:57.908293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def compute_binary_metrics(y_true, y_score, threshold=0.2):\n    y_true = np.asarray(y_true).astype(int)\n    y_score = np.asarray(y_score).astype(float)\n    y_pred = (y_score >= threshold).astype(int)\n\n    tn, fp, fn, tp = confusion_matrix(y_true, y_pred, labels=[0, 1]).ravel()\n\n    accuracy = (tp + tn) / (tp + tn + fp + fn) if (tp + tn + fp + fn) > 0 else np.nan\n    sensitivity = tp / (tp + fn) if (tp + fn) > 0 else np.nan\n    specificity = tn / (tn + fp) if (tn + fp) > 0 else np.nan\n\n    if len(np.unique(y_true)) < 2:\n        auc = np.nan\n    else:\n        auc = roc_auc_score(y_true, y_score)\n\n    return {\n        'accuracy': float(accuracy) if not np.isnan(accuracy) else np.nan,\n        'sensitivity': float(sensitivity) if not np.isnan(sensitivity) else np.nan,\n        'specificity': float(specificity) if not np.isnan(specificity) else np.nan,\n        'auc': float(auc) if not np.isnan(auc) else np.nan,\n        'tp': int(tp),\n        'tn': int(tn),\n        'fp': int(fp),\n        'fn': int(fn)\n    }\n\n\nptx_metrics = compute_binary_metrics(\n    y_true=eval_df['gt_pneumothorax'],\n    y_score=eval_df['ptx_confidence'],\n    threshold=PTX_THRESHOLD\n)\n\npe_metrics = compute_binary_metrics(\n    y_true=eval_df['gt_effusion'],\n    y_score=eval_df['effusion_confidence'],\n    threshold=EFFUSION_THRESHOLD\n)\n\nmetrics_df = pd.DataFrame([\n    {\n        'condition': 'Pneumothorax',\n        'threshold': PTX_THRESHOLD,\n        **ptx_metrics\n    },\n    {\n        'condition': 'Pleural Effusion',\n        'threshold': EFFUSION_THRESHOLD,\n        **pe_metrics\n    }\n])\n\nmetrics_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T09:35:57.909987Z","iopub.status.idle":"2026-04-13T09:35:57.910343Z","shell.execute_reply.started":"2026-04-13T09:35:57.910168Z","shell.execute_reply":"2026-04-13T09:35:57.910189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"metrics_df.to_csv('pe_ptx_metrics_train_split.csv', index=False)\neval_df.to_csv('pe_ptx_eval_details_train_split.csv', index=False)\n\nprint('Saved: pe_ptx_metrics_train_split.csv')\nprint('Saved: pe_ptx_eval_details_train_split.csv')\nprint('\\nKey metrics:')\nprint(metrics_df[['condition', 'threshold', 'accuracy', 'sensitivity', 'specificity', 'auc']])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T09:35:57.911677Z","iopub.status.idle":"2026-04-13T09:35:57.912014Z","shell.execute_reply.started":"2026-04-13T09:35:57.911823Z","shell.execute_reply":"2026-04-13T09:35:57.911842Z"}},"outputs":[],"execution_count":null}]}