{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport pydicom\n\n# file paths for the dataset\ntrain_path = '/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection/train'\ncsv_path = '/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection/train.csv'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:02.284079Z","iopub.execute_input":"2026-04-02T23:58:02.284864Z","iopub.status.idle":"2026-04-02T23:58:02.289389Z","shell.execute_reply.started":"2026-04-02T23:58:02.28483Z","shell.execute_reply":"2026-04-02T23:58:02.288352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# load the csv file into a dataframe\ndf = pd.read_csv(csv_path)\n\n# basic exploration\nprint(df.shape)           # how many rows and columns\nprint(df.head())          # first 5 rows\nprint(df.columns) # column names\nprint(df['class_name'].value_counts())  # how many of each class\nprint(df.isnull().sum())  # check for missing values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:02.290902Z","iopub.execute_input":"2026-04-02T23:58:02.291631Z","iopub.status.idle":"2026-04-02T23:58:02.413553Z","shell.execute_reply.started":"2026-04-02T23:58:02.291604Z","shell.execute_reply":"2026-04-02T23:58:02.41264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# function to load a dicom image and normalize it to 0-255\ndef load_dicom(image_id):\n    dicom = pydicom.dcmread(train_path + '/' + image_id + '.dicom')\n    img = dicom.pixel_array\n    img = (img - img.min()) / (img.max() - img.min()) * 255\n    return img.astype(np.uint8)\n\n# grab one normal and one abnormal image to display\nnormal_id = df[df['class_name'] == 'No finding']['image_id'].iloc[0]\nabnormal_id = df[df['class_name'] == 'Pleural effusion']['image_id'].iloc[0]\n\n# plot images as a test\nfig, axes = plt.subplots(1, 2, figsize=(12, 6))\naxes[0].imshow(load_dicom(normal_id), cmap='gray')\naxes[0].set_title('Normal')\naxes[1].imshow(load_dicom(abnormal_id), cmap='gray')\naxes[1].set_title('Pleural Effusion')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:02.415096Z","iopub.execute_input":"2026-04-02T23:58:02.415404Z","iopub.status.idle":"2026-04-02T23:58:07.265547Z","shell.execute_reply.started":"2026-04-02T23:58:02.415382Z","shell.execute_reply":"2026-04-02T23:58:07.264776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# resize and normalize all images for model training\n# we'll store them as numpy arrays to avoid reloading dicoms repeatedly\nIMG_SIZE = 256  # using 256 instead of 1024 to save memory\n\ndef preprocess_image(image_id):\n    dicom = pydicom.dcmread(train_path + '/' + image_id + '.dicom')\n    img = dicom.pixel_array\n    # normalize to 0-255\n    img = (img - img.min()) / (img.max() - img.min()) * 255\n    img = img.astype(np.uint8)\n    # resize to uniform size\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    return img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:07.266528Z","iopub.execute_input":"2026-04-02T23:58:07.266926Z","iopub.status.idle":"2026-04-02T23:58:07.271981Z","shell.execute_reply.started":"2026-04-02T23:58:07.2669Z","shell.execute_reply":"2026-04-02T23:58:07.271106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\n\n# get all unique image ids\nimage_ids = df['image_id'].unique()\nprint(f'Total unique images: {len(image_ids)}')\n\n# test the preprocessing on one image to make sure it works\ntest_img = preprocess_image(image_ids[0])\nprint(f'Image shape after preprocessing: {test_img.shape}')\nprint(f'Pixel range: {test_img.min()} to {test_img.max()}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:07.27378Z","iopub.execute_input":"2026-04-02T23:58:07.274366Z","iopub.status.idle":"2026-04-02T23:58:09.341377Z","shell.execute_reply.started":"2026-04-02T23:58:07.274342Z","shell.execute_reply":"2026-04-02T23:58:09.340503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# create a clean image-level dataframe\n# each image gets one row with a binary label (0 = normal, 1 = abnormal)\nimage_labels = df.groupby('image_id')['class_name'].apply(\n    lambda x: 0 if set(x) == {'No finding'} else 1\n).reset_index()\nimage_labels.columns = ['image_id', 'label']\n\nprint(image_labels['label'].value_counts())\nprint(f'\\n0 = Normal, 1 = Abnormal')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:09.342411Z","iopub.execute_input":"2026-04-02T23:58:09.342877Z","iopub.status.idle":"2026-04-02T23:58:09.565427Z","shell.execute_reply.started":"2026-04-02T23:58:09.342834Z","shell.execute_reply":"2026-04-02T23:58:09.564284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# sample 1500 from each class for a balanced 3000 image dataset\nnormal_ids = image_labels[image_labels['label'] == 0].sample(1500, random_state=42)\nabnormal_ids = image_labels[image_labels['label'] == 1].sample(1500, random_state=42)\nsample_df = pd.concat([normal_ids, abnormal_ids]).reset_index(drop=True)\n\nprint(sample_df['label'].value_counts())\n\ntrain_ids, val_ids, train_labels, val_labels = train_test_split(\n    sample_df['image_id'].values,\n    sample_df['label'].values,\n    test_size=0.2,\n    random_state=42,\n    stratify=sample_df['label'].values\n)\n\nprint(f'Training images: {len(train_ids)}')\nprint(f'Validation images: {len(val_ids)}')\n\n# recreate generators with smaller dataset\ntrain_gen = XrayGenerator(train_ids, train_labels, batch_size=32)\nval_gen = XrayGenerator(val_ids, val_labels, batch_size=32)\n\nprint(f'Training batches: {len(train_gen)}')\nprint(f'Validation batches: {len(val_gen)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:09.566553Z","iopub.execute_input":"2026-04-02T23:58:09.566842Z","iopub.status.idle":"2026-04-02T23:58:09.691448Z","shell.execute_reply.started":"2026-04-02T23:58:09.56681Z","shell.execute_reply":"2026-04-02T23:58:09.6907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\n# generator loads images in small batches instead of all at once\nclass XrayGenerator(tf.keras.utils.Sequence):\n    def __init__(self, image_ids, labels, batch_size=32):\n        self.image_ids = image_ids\n        self.labels = labels\n        self.batch_size = batch_size\n\n    def __len__(self):\n        # how many batches per epoch\n        return len(self.image_ids) // self.batch_size\n\n    def __getitem__(self, idx):\n        # grab a batch of image ids and labels\n        batch_ids = self.image_ids[idx * self.batch_size:(idx + 1) * self.batch_size]\n        batch_labels = self.labels[idx * self.batch_size:(idx + 1) * self.batch_size]\n\n        # load and preprocess each image in the batch\n        images = []\n        for image_id in batch_ids:\n            img = preprocess_image(image_id)\n            img = img / 255.0  # normalize to 0-1\n            img = np.expand_dims(img, axis=-1)  # add channel dimension for CNN\n            images.append(img)\n\n        return np.array(images), np.array(batch_labels)\n\n# create generators for training and validation\ntrain_gen = XrayGenerator(train_ids, train_labels, batch_size=32)\nval_gen = XrayGenerator(val_ids, val_labels, batch_size=32)\n\nprint(f'Training batches per epoch: {len(train_gen)}')\nprint(f'Validation batches per epoch: {len(val_gen)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:09.6926Z","iopub.execute_input":"2026-04-02T23:58:09.693117Z","iopub.status.idle":"2026-04-02T23:58:09.702231Z","shell.execute_reply.started":"2026-04-02T23:58:09.69309Z","shell.execute_reply":"2026-04-02T23:58:09.701288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# dice coefficient measures overlap between two radiologists' annotations\n# ranges from 0 (no overlap) to 1 (perfect overlap)\ndef dice_coefficient(box1, box2):\n    # find the intersection\n    x_min = max(box1[0], box2[0])\n    y_min = max(box1[1], box2[1])\n    x_max = min(box1[2], box2[2])\n    y_max = min(box1[3], box2[3])\n\n    intersection = max(0, x_max - x_min) * max(0, y_max - y_min)\n\n    # find the area of each box\n    area1 = (box1[2] - box1[0]) * (box1[3] - box1[1])\n    area2 = (box2[2] - box2[0]) * (box2[3] - box2[1])\n\n    # dice formula\n    if area1 + area2 == 0:\n        return 0\n    return (2 * intersection) / (area1 + area2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:09.703124Z","iopub.execute_input":"2026-04-02T23:58:09.703426Z","iopub.status.idle":"2026-04-02T23:58:09.720842Z","shell.execute_reply.started":"2026-04-02T23:58:09.703404Z","shell.execute_reply":"2026-04-02T23:58:09.720185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# find images annotated by more than one radiologist for comparison\nmulti_rad = df[df['class_name'] != 'No finding'].groupby(\n    ['image_id', 'class_name']\n).filter(lambda x: x['rad_id'].nunique() > 1)\n\nprint(f'Images with multiple radiologist annotations: {multi_rad[\"image_id\"].nunique()}')\nprint(f'Sample:\\n{multi_rad.head()}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:09.72186Z","iopub.execute_input":"2026-04-02T23:58:09.722232Z","iopub.status.idle":"2026-04-02T23:58:11.062651Z","shell.execute_reply.started":"2026-04-02T23:58:09.722207Z","shell.execute_reply":"2026-04-02T23:58:11.061679Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# calculate dice scores for each image/class with multiple annotations\ndice_scores = []\n\nfor (image_id, class_name), group in multi_rad.groupby(['image_id', 'class_name']):\n    rads = group['rad_id'].unique()\n    # compare first two radiologists' annotations\n    rad1 = group[group['rad_id'] == rads[0]][['x_min', 'y_min', 'x_max', 'y_max']].values[0]\n    rad2 = group[group['rad_id'] == rads[1]][['x_min', 'y_min', 'x_max', 'y_max']].values[0]\n    score = dice_coefficient(rad1, rad2)\n    dice_scores.append({'image_id': image_id, 'class_name': class_name, 'dice': score})\n\ndice_df = pd.DataFrame(dice_scores)\nprint(f'Average Dice Score: {dice_df[\"dice\"].mean():.4f}')\nprint(f'\\nDice by class:')\nprint(dice_df.groupby('class_name')['dice'].mean().sort_values(ascending=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:11.06495Z","iopub.execute_input":"2026-04-02T23:58:11.065288Z","iopub.status.idle":"2026-04-02T23:58:22.776685Z","shell.execute_reply.started":"2026-04-02T23:58:11.065259Z","shell.execute_reply":"2026-04-02T23:58:22.775989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# bar chart of dice scores by class\ndice_by_class = dice_df.groupby('class_name')['dice'].mean().sort_values(ascending=False)\n\nplt.figure(figsize=(12, 5))\nplt.bar(dice_by_class.index, dice_by_class.values, color='steelblue')\nplt.xticks(rotation=45, ha='right')\nplt.ylabel('Average Dice Score')\nplt.title('Inter-Radiologist Agreement by Class')\nplt.axhline(y=dice_by_class.mean(), color='red', linestyle='--', label=f'Mean: {dice_by_class.mean():.2f}')\nplt.legend()\nplt.tight_layout()\nplt.savefig('dice_scores.png', dpi=150)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:22.777633Z","iopub.execute_input":"2026-04-02T23:58:22.778386Z","iopub.status.idle":"2026-04-02T23:58:23.469357Z","shell.execute_reply.started":"2026-04-02T23:58:22.778359Z","shell.execute_reply":"2026-04-02T23:58:23.468572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# extract features directly from the csv bounding box data\n# no image loading needed at all\n\nabnormal = df[df['class_name'] != 'No finding'].copy()\n\n# size of each annotated region\nabnormal['box_width'] = abnormal['x_max'] - abnormal['x_min']\nabnormal['box_height'] = abnormal['y_max'] - abnormal['y_min']\nabnormal['box_area'] = abnormal['box_width'] * abnormal['box_height']\n\n# center of each box - tells us where in the image the finding is\nabnormal['box_center_x'] = (abnormal['x_min'] + abnormal['x_max']) / 2\nabnormal['box_center_y'] = (abnormal['y_min'] + abnormal['y_max']) / 2\n\nprint(abnormal[['class_name', 'box_area', 'box_width', 'box_height']].groupby('class_name').mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:23.470594Z","iopub.execute_input":"2026-04-02T23:58:23.470961Z","iopub.status.idle":"2026-04-02T23:58:23.50356Z","shell.execute_reply.started":"2026-04-02T23:58:23.470926Z","shell.execute_reply":"2026-04-02T23:58:23.502643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# visualize average finding size by class\navg_area = abnormal.groupby('class_name')['box_area'].mean().sort_values(ascending=False)\n\nplt.figure(figsize=(12, 5))\nplt.bar(avg_area.index, avg_area.values, color='steelblue')\nplt.xticks(rotation=45, ha='right')\nplt.ylabel('Average Bounding Box Area (pixels)')\nplt.title('Average Size of Findings by Class')\nplt.tight_layout()\nplt.savefig('feature_sizes.png', dpi=150)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:23.504705Z","iopub.execute_input":"2026-04-02T23:58:23.505038Z","iopub.status.idle":"2026-04-02T23:58:23.978022Z","shell.execute_reply.started":"2026-04-02T23:58:23.505002Z","shell.execute_reply":"2026-04-02T23:58:23.977114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# visualize where findings appear in the image (center x and y)\nfig, axes = plt.subplots(1, 2, figsize=(12, 5))\n\nfor class_name, group in abnormal.groupby('class_name'):\n    axes[0].scatter(group['box_center_x'].mean(), group['box_center_y'].mean(), label=class_name)\n    axes[1].scatter(group['box_area'].mean(), group['box_height'].mean(), label=class_name)\n\naxes[0].set_title('Average Finding Location in Image')\naxes[0].set_xlabel('Center X')\naxes[0].set_ylabel('Center Y')\naxes[0].invert_yaxis()  # image coordinates start from top\n\naxes[1].set_title('Average Area vs Height by Class')\naxes[1].set_xlabel('Box Area')\naxes[1].set_ylabel('Box Height')\naxes[1].ticklabel_format(style='plain', axis='x')  # stop scientific notation\nplt.setp(axes[1].get_xticklabels(), rotation=45, ha='right')  # angle the labels\n\nplt.legend(bbox_to_anchor=(1.05, 1), loc='upper left', fontsize=8)\nplt.tight_layout()\nplt.savefig('feature_locations.png', dpi=150)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:23.979413Z","iopub.execute_input":"2026-04-02T23:58:23.979727Z","iopub.status.idle":"2026-04-02T23:58:25.394408Z","shell.execute_reply.started":"2026-04-02T23:58:23.979701Z","shell.execute_reply":"2026-04-02T23:58:25.393586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# build a simple CNN for binary classification (normal vs abnormal)\nfrom tensorflow.keras import layers, models\n\nmodel = models.Sequential([\n    # first convolution block\n    layers.Conv2D(32, (3, 3), activation='relu', input_shape=(256, 256, 1)),\n    layers.MaxPooling2D(2, 2),\n    \n    # second convolution block\n    layers.Conv2D(64, (3, 3), activation='relu'),\n    layers.MaxPooling2D(2, 2),\n    \n    # third convolution block\n    layers.Conv2D(128, (3, 3), activation='relu'),\n    layers.MaxPooling2D(2, 2),\n    \n    # flatten and classify\n    layers.Flatten(),\n    layers.Dense(128, activation='relu'),\n    layers.Dropout(0.5),  # dropout helps prevent overfitting\n    layers.Dense(1, activation='sigmoid')  # sigmoid for binary output\n])\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:25.395488Z","iopub.execute_input":"2026-04-02T23:58:25.395801Z","iopub.status.idle":"2026-04-02T23:58:28.406397Z","shell.execute_reply.started":"2026-04-02T23:58:25.395766Z","shell.execute_reply":"2026-04-02T23:58:28.405548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# handle class imbalance by weighting the minority class higher\n# 10606 normal vs 4394 abnormal - we need to account for this\ntotal = len(train_labels)\nnormal_weight = total / (2 * np.sum(train_labels == 0))\nabnormal_weight = total / (2 * np.sum(train_labels == 1))\nclass_weights = {0: normal_weight, 1: abnormal_weight}\nprint(f'Class weights: {class_weights}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:28.40742Z","iopub.execute_input":"2026-04-02T23:58:28.407727Z","iopub.status.idle":"2026-04-02T23:58:28.412948Z","shell.execute_reply.started":"2026-04-02T23:58:28.407695Z","shell.execute_reply":"2026-04-02T23:58:28.412013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nfrom tqdm.notebook import tqdm\n\n# convert dicoms to jpgs once so training is much faster\njpg_path = '/kaggle/working/images'\nos.makedirs(jpg_path, exist_ok=True)\n\nfor image_id in tqdm(sample_df['image_id'].values):\n    dicom = pydicom.dcmread(train_path + '/' + image_id + '.dicom')\n    img = dicom.pixel_array\n    img = (img - img.min()) / (img.max() - img.min()) * 255\n    img = img.astype(np.uint8)\n    img = cv2.resize(img, (256, 256))\n    cv2.imwrite(jpg_path + '/' + image_id + '.jpg', img)\n\nprint('Done converting images')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T23:58:28.413977Z","iopub.execute_input":"2026-04-02T23:58:28.414297Z","iopub.status.idle":"2026-04-03T00:46:03.71583Z","shell.execute_reply.started":"2026-04-02T23:58:28.414263Z","shell.execute_reply":"2026-04-03T00:46:03.714849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"files = os.listdir(jpg_path)\nprint(f'JPG files found: {len(files)}')\nprint(f'Example: {files[0]}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-03T00:46:03.717017Z","iopub.execute_input":"2026-04-03T00:46:03.717393Z","iopub.status.idle":"2026-04-03T00:46:03.724872Z","shell.execute_reply.started":"2026-04-03T00:46:03.717367Z","shell.execute_reply":"2026-04-03T00:46:03.724192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class XrayGenerator(tf.keras.utils.Sequence):\n    def __init__(self, image_ids, labels, batch_size=32):\n        self.image_ids = image_ids\n        self.labels = labels\n        self.batch_size = batch_size\n\n    def __len__(self):\n        return len(self.image_ids) // self.batch_size\n\n    def __getitem__(self, idx):\n        batch_ids = self.image_ids[idx * self.batch_size:(idx + 1) * self.batch_size]\n        batch_labels = self.labels[idx * self.batch_size:(idx + 1) * self.batch_size]\n\n        images = []\n        for image_id in batch_ids:\n            # read jpg instead of dicom - much faster\n            img = cv2.imread(jpg_path + '/' + image_id + '.jpg', cv2.IMREAD_GRAYSCALE)\n            img = img / 255.0\n            img = np.expand_dims(img, axis=-1)\n            images.append(img)\n\n        return np.array(images), np.array(batch_labels)\n\ntrain_gen = XrayGenerator(train_ids, train_labels, batch_size=32)\nval_gen = XrayGenerator(val_ids, val_labels, batch_size=32)\nprint('Generators ready')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-03T00:46:03.726615Z","iopub.execute_input":"2026-04-03T00:46:03.727374Z","iopub.status.idle":"2026-04-03T00:46:03.736628Z","shell.execute_reply.started":"2026-04-03T00:46:03.727348Z","shell.execute_reply":"2026-04-03T00:46:03.735817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# compile the model\nmodel.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['accuracy']\n)\n\n# train the model\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=10,\n    class_weight=class_weights\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-03T00:46:03.737643Z","iopub.execute_input":"2026-04-03T00:46:03.73794Z","iopub.status.idle":"2026-04-03T00:47:12.148367Z","shell.execute_reply.started":"2026-04-03T00:46:03.737917Z","shell.execute_reply":"2026-04-03T00:47:12.147411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot accuracy and loss over epochs\nfig, axes = plt.subplots(1, 2, figsize=(12, 4))\n\n# accuracy\naxes[0].plot(history.history['accuracy'], label='Training')\naxes[0].plot(history.history['val_accuracy'], label='Validation')\naxes[0].set_title('Model Accuracy')\naxes[0].set_xlabel('Epoch')\naxes[0].set_ylabel('Accuracy')\naxes[0].legend()\n\n# loss\naxes[1].plot(history.history['loss'], label='Training')\naxes[1].plot(history.history['val_loss'], label='Validation')\naxes[1].set_title('Model Loss')\naxes[1].set_xlabel('Epoch')\naxes[1].set_ylabel('Loss')\naxes[1].legend()\n\nplt.tight_layout()\nplt.savefig('training_history.png', dpi=150)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-03T01:23:05.121854Z","iopub.execute_input":"2026-04-03T01:23:05.122525Z","iopub.status.idle":"2026-04-03T01:23:05.658199Z","shell.execute_reply.started":"2026-04-03T01:23:05.122487Z","shell.execute_reply":"2026-04-03T01:23:05.657331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, confusion_matrix\nimport seaborn as sns\n\n# get predictions on validation set\nval_preds = model.predict(val_gen)\nval_preds_binary = (val_preds > 0.5).astype(int).flatten()\n\n# only use labels that match the generator output size\nval_labels_trimmed = val_labels[:len(val_preds_binary)]\n\n# calculate metrics\naccuracy = accuracy_score(val_labels_trimmed, val_preds_binary)\nprecision = precision_score(val_labels_trimmed, val_preds_binary)\nrecall = recall_score(val_labels_trimmed, val_preds_binary)\nf1 = f1_score(val_labels_trimmed, val_preds_binary)\n\nprint(f'Accuracy:  {accuracy:.4f}')\nprint(f'Precision: {precision:.4f}')\nprint(f'Recall:    {recall:.4f}')\nprint(f'F1 Score:  {f1:.4f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-03T01:24:00.73896Z","iopub.execute_input":"2026-04-03T01:24:00.739717Z","iopub.status.idle":"2026-04-03T01:24:01.988566Z","shell.execute_reply.started":"2026-04-03T01:24:00.739683Z","shell.execute_reply":"2026-04-03T01:24:01.987635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# confusion matrix\ncm = confusion_matrix(val_labels_trimmed, val_preds_binary)\n\nplt.figure(figsize=(6, 5))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n            xticklabels=['Normal', 'Abnormal'],\n            yticklabels=['Normal', 'Abnormal'])\nplt.ylabel('Actual')\nplt.xlabel('Predicted')\nplt.title('Confusion Matrix')\nplt.savefig('confusion_matrix.png', dpi=150)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-03T01:24:04.928647Z","iopub.execute_input":"2026-04-03T01:24:04.929353Z","iopub.status.idle":"2026-04-03T01:24:05.185578Z","shell.execute_reply.started":"2026-04-03T01:24:04.929321Z","shell.execute_reply":"2026-04-03T01:24:05.184727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import precision_recall_curve\n\n# precision recall curve\nprecision_vals, recall_vals, thresholds = precision_recall_curve(val_labels_trimmed, val_preds[:len(val_labels_trimmed)])\n\nplt.figure(figsize=(7, 5))\nplt.plot(recall_vals, precision_vals, color='steelblue')\nplt.xlabel('Recall')\nplt.ylabel('Precision')\nplt.title('Precision-Recall Curve')\nplt.savefig('precision_recall.png', dpi=150)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-03T01:24:08.631885Z","iopub.execute_input":"2026-04-03T01:24:08.632657Z","iopub.status.idle":"2026-04-03T01:24:08.855843Z","shell.execute_reply.started":"2026-04-03T01:24:08.632611Z","shell.execute_reply":"2026-04-03T01:24:08.85517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# find misclassified images for error analysis\nval_ids_trimmed = val_ids[:len(val_preds_binary)]\n\n# get false negatives and false positives\nfalse_negatives = [val_ids_trimmed[i] for i in range(len(val_preds_binary)) \n                   if val_preds_binary[i] == 0 and val_labels_trimmed[i] == 1]\nfalse_positives = [val_ids_trimmed[i] for i in range(len(val_preds_binary)) \n                   if val_preds_binary[i] == 1 and val_labels_trimmed[i] == 0]\n\nprint(f'False negatives: {len(false_negatives)}')\nprint(f'False positives: {len(false_positives)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-03T01:27:01.858576Z","iopub.execute_input":"2026-04-03T01:27:01.859477Z","iopub.status.idle":"2026-04-03T01:27:01.865762Z","shell.execute_reply.started":"2026-04-03T01:27:01.859442Z","shell.execute_reply":"2026-04-03T01:27:01.865084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# visualize some misclassified images\nfig, axes = plt.subplots(2, 4, figsize=(14, 7))\n\nfor i, img_id in enumerate(false_negatives[:4]):\n    img = cv2.imread(jpg_path + '/' + img_id + '.jpg', cv2.IMREAD_GRAYSCALE)\n    axes[0, i].imshow(img, cmap='gray')\n    axes[0, i].set_title('False Negative')\n    axes[0, i].axis('off')\n\nfor i, img_id in enumerate(false_positives[:4]):\n    img = cv2.imread(jpg_path + '/' + img_id + '.jpg', cv2.IMREAD_GRAYSCALE)\n    axes[1, i].imshow(img, cmap='gray')\n    axes[1, i].set_title('False Positive')\n    axes[1, i].axis('off')\n\nplt.suptitle('Misclassified Images')\nplt.tight_layout()\nplt.savefig('misclassified.png', dpi=150)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-03T01:27:07.527933Z","iopub.execute_input":"2026-04-03T01:27:07.528841Z","iopub.status.idle":"2026-04-03T01:27:09.177962Z","shell.execute_reply.started":"2026-04-03T01:27:07.528805Z","shell.execute_reply":"2026-04-03T01:27:09.177185Z"}},"outputs":[],"execution_count":null}]}