{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":13451,"databundleVersionId":1188070,"sourceType":"competition"},{"sourceId":13402655,"sourceType":"datasetVersion","datasetId":8505503},{"sourceId":609905,"sourceType":"modelInstanceVersion","modelInstanceId":457976,"modelId":473896}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"ls /kaggle/input/history","metadata":{"_uuid":"e96bcf7e-5b7e-4477-96f4-2466f489b536","_cell_guid":"9dd23315-824b-4e53-a034-84feb7dcb4cc","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-10-16T09:14:06.523963Z","iopub.execute_input":"2025-10-16T09:14:06.524245Z","iopub.status.idle":"2025-10-16T09:14:06.648384Z","shell.execute_reply.started":"2025-10-16T09:14:06.524223Z","shell.execute_reply":"2025-10-16T09:14:06.647525Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rm -rf /kaggle/working/data_part_2","metadata":{"_uuid":"2be6fd59-0c1e-4c9d-9f5d-851ec67ada7e","_cell_guid":"a27fabe3-1f48-4d90-9e99-a3da853c959e","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-10-15T18:51:40.463517Z","iopub.execute_input":"2025-10-15T18:51:40.464448Z","iopub.status.idle":"2025-10-15T18:51:41.324162Z","shell.execute_reply.started":"2025-10-15T18:51:40.464416Z","shell.execute_reply":"2025-10-15T18:51:41.323208Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport random, shutil\nfrom pathlib import Path\n\n# Set a fixed seed for reproducibility\nrandom.seed(42)\n\n# Paths\nSRC_DIR = Path(\"/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train\")\nCSV_PATH = \"/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train.csv\"\nDST_DIR = Path(\"/kaggle/working/data_30k\")\nDST_DIR.mkdir(exist_ok=True)\n\n# 1. Read labels and prepare data\ndf = pd.read_csv(CSV_PATH)\ndf['Subtype'] = df['ID'].apply(lambda x: x.split('_')[-1])\ndf['ImageID'] = df['ID'].apply(lambda x: \"_\".join(x.split('_')[:2]))\n\nsubtypes = ['epidural','intraparenchymal','intraventricular','subarachnoid','subdural']\n\n# Pivot table to get one row per image with its labels\ndf_pivot = df.pivot_table(index='ImageID', columns='Subtype', values='Label', aggfunc='max').fillna(0).reset_index()\ndf_pivot['any'] = df_pivot[subtypes].max(axis=1)\n\n# 2. Define number of images to sample per class\nsubtype_counts = {\n    'epidural': 3000,\n    'intraparenchymal': 4000,\n    'intraventricular': 4000,\n    'subarachnoid': 4000,\n    'subdural': 4000\n}\nnegative_count = 11000\n\n# 3. Select positive images (per subtype)\nselected_ids = set()\nselected_by_class = {}\n\nfor subtype, num in subtype_counts.items():\n    ids = df_pivot[df_pivot[subtype] == 1]['ImageID'].tolist()\n    chosen = random.sample(ids, min(len(ids), num))\n    selected_ids.update(chosen)\n    selected_by_class[subtype] = chosen\n    print(f\"Selected {len(chosen)} {subtype} images\")\n\n# 4. Select negative images\nneg_ids = df_pivot[df_pivot['any'] == 0]['ImageID'].tolist()\nchosen_neg = random.sample(neg_ids, negative_count)\nselected_ids.update(chosen_neg)\nselected_by_class['negative'] = chosen_neg\nprint(f\"Selected {len(chosen_neg)} negative images\")\n\nprint(f\"Total images selected: {len(selected_ids)}\")\n\n# 5. Save labels to CSV\ndf_selected = df[df['ImageID'].isin(selected_ids)]\ndf_selected.to_csv(\"/kaggle/working/data_30k_labels.csv\", index=False)\nprint(\"Labels saved to data_30k_labels.csv\")\n\n# 6. Copy images by class with progress\nfor subtype, ids in selected_by_class.items():\n    total = len(ids)\n    print(f\"\\nCopying {subtype} ({total} images)...\")\n    for i, img_id in enumerate(ids, 1):\n        src = SRC_DIR / f\"{img_id}.dcm\"\n        if src.exists():\n            shutil.copy(src, DST_DIR / f\"{img_id}.dcm\")\n        if i % 1000 == 0 or i == total:\n            print(f\"   Copied {i}/{total} images for {subtype}\")\n\nprint(\"All images copied successfully!\")","metadata":{"_uuid":"aa1b2321-dc19-46ac-b68a-2ad43e25521a","_cell_guid":"f45993c4-ebed-4cdd-8a34-3c8c14a44259","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-10-16T05:34:21.858947Z","iopub.execute_input":"2025-10-16T05:34:21.859147Z","iopub.status.idle":"2025-10-16T05:41:27.389561Z","shell.execute_reply.started":"2025-10-16T05:34:21.85913Z","shell.execute_reply":"2025-10-16T05:41:27.388905Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport pydicom\n\nfrom os import listdir\nfrom os.path import isfile, join","metadata":{"_uuid":"4404ade7-7b2f-4e1e-b9b1-1ad42b05d70d","_cell_guid":"1b83f77f-db5a-4b41-84b0-0530857a012d","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-10-16T05:45:22.339364Z","iopub.execute_input":"2025-10-16T05:45:22.339956Z","iopub.status.idle":"2025-10-16T05:45:23.903654Z","shell.execute_reply.started":"2025-10-16T05:45:22.339932Z","shell.execute_reply":"2025-10-16T05:45:23.903108Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load data\nCSV_PATH = '/kaggle/working/data_30k_labels.csv'\nIMAGE_DIR = '/kaggle/working/data_30k/'\n# Read file\ntrain = pd.read_csv(CSV_PATH)\nprint(\"Data shape:\", train.shape)\ntrain.head()\n\n# Read number of image\ntrain_images = [f for f in listdir(IMAGE_DIR) if isfile(join(IMAGE_DIR, f))]\nprint('Number of train images:', len(train_images))\n\n# Add Sub_type và PatientID\ntrain['Sub_type'] = train['ID'].str.split(\"_\", n=3, expand=True)[2]\ntrain['PatientID'] = train['ID'].str.split(\"_\", n=3, expand=True)[1]\n\nprint(train.head())","metadata":{"_uuid":"f7165cbb-bbd3-485e-947b-8783d8bf477e","_cell_guid":"3efba5ed-5b7a-4894-8b5d-942f9461c49f","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-10-16T05:46:10.536091Z","iopub.execute_input":"2025-10-16T05:46:10.536765Z","iopub.status.idle":"2025-10-16T05:46:11.790963Z","shell.execute_reply.started":"2025-10-16T05:46:10.536737Z","shell.execute_reply":"2025-10-16T05:46:11.790281Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Label distribution\nplt.figure(figsize=(6,4))\nsns.countplot(x='Label', data=train)\nplt.title(\"Label Distribution (0 = negative, 1 = positive)\")\nplt.show()\n\nprint(train['Label'].value_counts())\n\n# Distribute by subtype\ngbSub = train.groupby('Sub_type').sum(numeric_only=True)\nplt.figure(figsize=(10,6))\nsns.barplot(y=gbSub.index, x=gbSub.Label, palette=\"deep\")\nplt.title(\"Total Positive Labels by Subtype\")\nplt.show()\n\nfig=plt.figure(figsize=(10, 8))\nsns.countplot(x=\"Sub_type\", hue=\"Label\", data=train)\nplt.title(\"Total Images by Subtype\")\nplt.xticks(rotation=15)\nplt.show()","metadata":{"_uuid":"7eacfc3e-6777-4993-ae2e-7232db6590aa","_cell_guid":"7de229e4-371e-44dc-b95c-4895ef7f4a77","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-10-16T05:46:14.65614Z","iopub.execute_input":"2025-10-16T05:46:14.6564Z","iopub.status.idle":"2025-10-16T05:46:15.379403Z","shell.execute_reply.started":"2025-10-16T05:46:14.656381Z","shell.execute_reply":"2025-10-16T05:46:15.378604Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pydicom import dcmread\n# --- Read DICOM and window image ---\ndef window_image(img, window_center, window_width, intercept, slope):\n    img = (img * slope + intercept)\n    img_min = window_center - window_width // 2\n    img_max = window_center + window_width // 2\n    img = img.clip(img_min, img_max)\n    return img\n\n\ndef get_first_of_dicom_field_as_int(x):\n    if isinstance(x, pydicom.multival.MultiValue):\n        return int(x[0])\n    return int(x)\n\n\ndef get_windowing(data):\n    dicom_fields = [\n        data[('0028', '1050')].value,  # window center\n        data[('0028', '1051')].value,  # window width\n        data[('0028', '1052')].value,  # intercept\n        data[('0028', '1053')].value,  # slope\n    ]\n    return [get_first_of_dicom_field_as_int(x) for x in dicom_fields]\n\n\n# --- Show DICOM images by subtype ---\ndef view_images(patient_ids, title='', n_rows=2, n_cols=5):\n    fig, axs = plt.subplots(n_rows, n_cols, figsize=(15, 6))\n    axs = axs.flatten()\n\n    for idx, pid in enumerate(patient_ids[:n_rows * n_cols]):\n        file_path = os.path.join(IMAGE_DIR, f'ID_{pid}.dcm')\n        if not os.path.exists(file_path):\n            continue\n\n        try:\n            data = dcmread(file_path)  # ✅ updated function\n            image = data.pixel_array\n            window_center, window_width, intercept, slope = get_windowing(data)\n            image_windowed = window_image(image, window_center, window_width, intercept, slope)\n\n            axs[idx].imshow(image_windowed, cmap=plt.cm.bone)\n            axs[idx].axis('off')\n        except Exception as e:\n            print(f\"Error reading {file_path}: {e}\")\n\n    plt.suptitle(title)\n    plt.tight_layout()\n    plt.show()","metadata":{"_uuid":"6cd6135b-e81b-4fb6-abab-1359478cf900","_cell_guid":"dfa5fd31-0cdd-4748-9d05-8e729f913e97","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-10-16T05:46:20.755683Z","iopub.execute_input":"2025-10-16T05:46:20.756236Z","iopub.status.idle":"2025-10-16T05:46:20.764483Z","shell.execute_reply.started":"2025-10-16T05:46:20.756211Z","shell.execute_reply":"2025-10-16T05:46:20.763719Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Display images by type of bleeding\nfor subtype in ['epidural','intraparenchymal','intraventricular','subarachnoid','subdural']:\n    ids = train[(train['Sub_type'] == subtype) & (train['Label'] == 1)]['PatientID'].values\n    print(f\"Showing {subtype} - {len(ids)} positive images available\")\n    if len(ids) > 0:\n        view_images(ids, title=f'Images of hemorrhage: {subtype}')","metadata":{"_uuid":"ab95783b-bdb0-4792-a903-bc2569760741","_cell_guid":"057ed222-7b8b-4ac4-8775-4f6ab68bd4d0","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-10-15T21:10:53.744214Z","iopub.execute_input":"2025-10-15T21:10:53.744903Z","iopub.status.idle":"2025-10-15T21:10:56.579928Z","shell.execute_reply.started":"2025-10-15T21:10:53.744875Z","shell.execute_reply":"2025-10-15T21:10:56.579129Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"1. Đọc & tiền xử lý ảnh DICOM (windowing brain + subdural + soft tissue)\n2. Data generator\n3. Weighted BCE loss + weighted metric\n4. Huấn luyện mô hình InceptionV3 pretrained\n5. Chia train/val/test\n6. Lưu kết quả dự đoán","metadata":{"_uuid":"b4de250f-5f6c-424b-9d16-d6aeae38724a","_cell_guid":"66346e9c-ef91-4938-8141-e39c85b0fe80","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pydicom\nimport cv2\nimport os\nimport matplotlib.pyplot as plt\n\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\nfrom tensorflow.keras.applications import InceptionV3\nfrom tensorflow.keras import backend as K\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\nimport keras\nimport albumentations as A","metadata":{"_uuid":"c5609731-9454-4fc7-8486-c9feae13d3e1","_cell_guid":"7f506d46-e8b5-4ab1-b03b-e713a35e11d5","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-10-16T05:49:43.979078Z","iopub.execute_input":"2025-10-16T05:49:43.980063Z","iopub.status.idle":"2025-10-16T05:49:43.984765Z","shell.execute_reply.started":"2025-10-16T05:49:43.980027Z","shell.execute_reply":"2025-10-16T05:49:43.983895Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CSV_PATH = '/kaggle/working/data_30k_labels.csv'\nIMAGE_DIR = '/kaggle/working/data_30k/'\n\ndf_raw = pd.read_csv(CSV_PATH)\n\n# Separate Image ID và Diagnosis\ndf_raw[\"Image\"] = df_raw[\"ID\"].str.slice(stop=12)\ndf_raw[\"Diagnosis\"] = df_raw[\"ID\"].str.slice(start=13)\n\n# Change from long to wide format (1 dòng / ảnh, 6 cột nhãn)\ndf = df_raw.loc[:, [\"Label\", \"Diagnosis\", \"Image\"]]\ndf = df.set_index(['Image', 'Diagnosis']).unstack(level=-1)\n\nprint(\"Shape of train data:\", df.shape)\ndf.head()","metadata":{"_uuid":"8ab030dc-853d-421d-893f-dd11c6055f8e","_cell_guid":"beed9a91-b335-43fc-a2bc-a06d7488b046","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-16T05:49:46.260912Z","iopub.execute_input":"2025-10-16T05:49:46.261201Z","iopub.status.idle":"2025-10-16T05:49:46.77995Z","shell.execute_reply.started":"2025-10-16T05:49:46.26118Z","shell.execute_reply":"2025-10-16T05:49:46.779313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def correct_dcm(dcm):\n    x = dcm.pixel_array + 1000\n    px_mode = 4096\n    x[x>=px_mode] = x[x>=px_mode] - px_mode\n    dcm.PixelData = x.tobytes()\n    dcm.RescaleIntercept = -1000\n\ndef window_image(dcm, window_center, window_width):\n    if (dcm.BitsStored == 12) and (dcm.PixelRepresentation == 0) and (int(dcm.RescaleIntercept) > -100):\n        correct_dcm(dcm)\n    img = dcm.pixel_array * dcm.RescaleSlope + dcm.RescaleIntercept\n    img_min = window_center - window_width // 2\n    img_max = window_center + window_width // 2\n    img = np.clip(img, img_min, img_max)\n    return img\n\ndef bsb_window(dcm):\n    brain_img = window_image(dcm, 40, 80)\n    subdural_img = window_image(dcm, 80, 200)\n    soft_img = window_image(dcm, 40, 380)\n\n    brain_img = (brain_img - 0) / 80\n    subdural_img = (subdural_img - (-20)) / 200\n    soft_img = (soft_img - (-150)) / 380\n\n    return np.array([brain_img, subdural_img, soft_img]).transpose(1,2,0)\n\ndef _read(path, desired_size):\n    dcm = pydicom.dcmread(path)\n    try:\n        img = bsb_window(dcm)\n    except:\n        img = np.zeros(desired_size)\n    img = cv2.resize(img, desired_size[:2], interpolation=cv2.INTER_LINEAR)\n    return img","metadata":{"_uuid":"11290fe9-fd8c-41dd-9b3f-759aacc346b7","_cell_guid":"4b580aa2-2a75-40df-b6c6-6f5c4d65962c","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-16T05:49:51.508318Z","iopub.execute_input":"2025-10-16T05:49:51.508915Z","iopub.status.idle":"2025-10-16T05:49:51.51596Z","shell.execute_reply.started":"2025-10-16T05:49:51.508855Z","shell.execute_reply":"2025-10-16T05:49:51.515122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DataGenerator(keras.utils.Sequence):\n    def __init__(\n        self,\n        list_IDs,\n        labels=None,\n        batch_size=16,\n        img_size=(224, 224, 3),\n        img_dir=\"/kaggle/working/data_30k/\",\n        shuffle=True,\n        augment=False,\n    ):\n        self.list_IDs = list_IDs  # list of image IDs (e.g., \"ID_12345\")\n        self.labels = labels      # DataFrame with index = ImageID\n        self.batch_size = batch_size\n        self.img_size = img_size\n        self.img_dir = img_dir\n        self.shuffle = shuffle\n        self.augment = augment\n        self.on_epoch_end()\n        self.transform = A.Compose([\n            A.HorizontalFlip(p=0.5),\n            A.ShiftScaleRotate(shift_limit=0.1, scale_limit=0.1, rotate_limit=15, p=0.5),\n            A.RandomBrightnessContrast(brightness_limit=0.2, contrast_limit=0.2, p=0.5),\n            A.GaussNoise(p=0.3),\n        ]) if augment else None\n\n    def __len__(self):\n        \"\"\"Number of batches per epoch\"\"\"\n        return int(np.ceil(len(self.indices) / self.batch_size))\n\n    def __getitem__(self, index):\n        \"\"\"Generate one batch\"\"\"\n        idxs = self.indices[index * self.batch_size:(index + 1) * self.batch_size]\n        list_IDs_temp = [self.list_IDs[k] for k in idxs]\n\n        X, Y = self.__data_generation(list_IDs_temp)\n\n        # Training / Validation\n        if self.labels is not None:\n            return X, Y\n        # Test mode (no labels)\n        return X\n\n    def on_epoch_end(self):\n        \"\"\"Updates indices after each epoch\"\"\"\n        self.indices = np.arange(len(self.list_IDs))\n        if self.shuffle:\n            np.random.shuffle(self.indices)\n\n    def __data_generation(self, list_IDs_temp):\n        \"\"\"Loads and processes batch of images\"\"\"\n        X = np.empty((len(list_IDs_temp), *self.img_size), dtype=np.float32)\n        Y = None\n\n        if self.labels is not None:\n            Y = np.empty((len(list_IDs_temp), 6), dtype=np.float32)\n\n        for i, ID in enumerate(list_IDs_temp):\n            dcm_path = f\"{self.img_dir}{ID}.dcm\"\n            img = self._read_dicom(dcm_path)\n\n            if self.augment:\n                img = self._augment(img)\n\n            X[i,] = img\n\n            if self.labels is not None:\n                Y[i,] = self.labels.loc[ID].values.astype(np.float32)\n\n        return X, Y\n\n    def _read_dicom(self, path):\n        \"\"\"Read and preprocess DICOM -> RGB np.array\"\"\"\n        return _read(path, self.img_size)\n\n    def _augment(self, img):\n        \"\"\"Apply albumentations augmentation\"\"\"\n        if self.transform is not None:\n            img = self.transform(image=img)['image']\n        return img","metadata":{"_uuid":"e8e0bef5-01f6-4573-8e55-6543756aabc7","_cell_guid":"d8074e4a-8785-42be-a2b8-0a1b9300b701","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-10-16T05:49:54.40512Z","iopub.execute_input":"2025-10-16T05:49:54.40561Z","iopub.status.idle":"2025-10-16T05:49:54.41544Z","shell.execute_reply.started":"2025-10-16T05:49:54.405588Z","shell.execute_reply":"2025-10-16T05:49:54.414661Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def weighted_log_loss(y_true, y_pred):\n    class_weights = np.array([1.,2., 1., 1., 1., 1.])\n    eps = K.epsilon()\n    y_pred = K.clip(y_pred, eps, 1.0-eps)\n    out = -(y_true * K.log(y_pred) * class_weights + (1.0 - y_true) * K.log(1.0 - y_pred) * class_weights)\n    return K.mean(out, axis=-1)\n\ndef _normalized_weighted_average(arr, weights=None):\n    if weights is not None:\n        scl = K.sum(weights)\n        weights = K.expand_dims(weights, axis=1)\n        return K.sum(K.dot(arr, weights), axis=1) / scl\n    return K.mean(arr, axis=1)\n\ndef weighted_loss(y_true, y_pred):\n    class_weights = K.constant([1., 2., 1., 1., 1., 1.])\n    eps = K.epsilon()\n    y_pred = K.clip(y_pred, eps, 1.0-eps)\n    loss = -(y_true * K.log(y_pred) + (1.0 - y_true) * K.log(1.0 - y_pred))\n    loss_samples = _normalized_weighted_average(loss, class_weights)\n    return K.mean(loss_samples)","metadata":{"_uuid":"a374fb65-7111-42ef-9b55-20091e2615b8","_cell_guid":"9c2d3686-73b1-44de-863d-100244692be7","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-16T05:49:57.516806Z","iopub.execute_input":"2025-10-16T05:49:57.517577Z","iopub.status.idle":"2025-10-16T05:49:57.523538Z","shell.execute_reply.started":"2025-10-16T05:49:57.517551Z","shell.execute_reply":"2025-10-16T05:49:57.522835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class HemorrhageClassifier:\n    def __init__(self, engine, input_dims, batch_size=16, num_epochs=5, learning_rate=5e-5, \n                 decay_rate=0.8, decay_steps=1, weights=\"imagenet\", verbose=1, model_name=\"model\"):\n        self.engine = engine\n        self.input_dims = input_dims\n        self.batch_size = batch_size\n        self.num_epochs = num_epochs\n        self.learning_rate = learning_rate\n        self.decay_rate = decay_rate\n        self.decay_steps = decay_steps\n        self.weights = weights\n        self.verbose = verbose\n        self.model_name = model_name\n        self._build()\n\n    def _build(self):\n        engine = self.engine(\n            include_top=False,\n            weights=self.weights,\n            input_shape=self.input_dims,\n        )\n        x = keras.layers.GlobalAveragePooling2D(name='avg_pool')(engine.output)\n        x = keras.layers.Dropout(0.5)(x)\n        out = keras.layers.Dense(6, activation=\"sigmoid\", name='dense_output')(x)\n    \n        self.model = keras.models.Model(inputs=engine.input, outputs=out)\n        self.model.compile(\n            loss=weighted_loss,\n            optimizer=keras.optimizers.Adam(learning_rate=self.learning_rate),\n            metrics=[\n                keras.metrics.BinaryAccuracy(name='binary_accuracy'),\n                keras.metrics.AUC(multi_label=True, name='auc'),\n            ]\n        )\n\n    def fit_and_predict(self, train_df, valid_df, test_df):\n        reduce_lr = ReduceLROnPlateau(\n            monitor='val_loss',\n            factor=0.5,\n            patience=3,\n            min_lr=1e-7,\n            verbose=1\n        )\n        early_stop = EarlyStopping(monitor='val_loss', patience=3, restore_best_weights=True)\n        checkpoint_path = f\"/kaggle/working/best_{self.model_name}.weights.h5\"\n        checkpoint = ModelCheckpoint(\n            checkpoint_path,\n            monitor='val_loss',\n            save_best_only=True,\n            save_weights_only=True,\n            verbose=1\n        )\n        history = self.model.fit(\n            DataGenerator(train_df.index, train_df, self.batch_size, self.input_dims),\n            epochs=self.num_epochs,\n            validation_data=DataGenerator(valid_df.index, valid_df, self.batch_size, self.input_dims),\n            verbose=self.verbose,\n            callbacks=[reduce_lr, early_stop, checkpoint]\n        )\n\n        self.model.load_weights(checkpoint_path)\n        preds = self.model.predict(DataGenerator(test_df.index, None, self.batch_size, self.input_dims), verbose=1)\n        self.history = history\n        return preds\n\n    def save(self, path):\n        self.model.save_weights(path)\n\n    def load(self, path):\n        self.model.load_weights(path)","metadata":{"_uuid":"2b010a19-ecb4-4b29-9697-306111edc066","_cell_guid":"431f727d-81e5-4d57-8366-09e225b1492f","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-10-16T05:50:00.63736Z","iopub.execute_input":"2025-10-16T05:50:00.637883Z","iopub.status.idle":"2025-10-16T05:50:00.646561Z","shell.execute_reply.started":"2025-10-16T05:50:00.637844Z","shell.execute_reply":"2025-10-16T05:50:00.645899Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.applications import InceptionV3, ResNet50, DenseNet121","metadata":{"_uuid":"bb0a7760-9746-4714-ae1e-1d573ce39d31","_cell_guid":"f5aec0a5-debb-483f-adc2-7cf25d860cef","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-10-16T05:50:03.213377Z","iopub.execute_input":"2025-10-16T05:50:03.213639Z","iopub.status.idle":"2025-10-16T05:50:03.217556Z","shell.execute_reply.started":"2025-10-16T05:50:03.213619Z","shell.execute_reply":"2025-10-16T05:50:03.216902Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inception_model = HemorrhageClassifier(\n    engine=InceptionV3,\n    input_dims=(299, 299, 3),\n    batch_size=16,\n    num_epochs=20,\n    model_name=\"inceptionv3\"\n)\n\nresnet_model = HemorrhageClassifier(\n    engine=ResNet50,\n    input_dims=(299, 299, 3),\n    batch_size=16,\n    num_epochs=20,\n    model_name=\"resnet50\"\n)\n\ndense_model = HemorrhageClassifier(\n    engine=DenseNet121,\n    input_dims=(224, 224, 3),\n    batch_size=16,\n    num_epochs=20,\n    model_name=\"densenet121\"\n)","metadata":{"_uuid":"f9efaa2d-8184-476b-8d84-6651ab2354e2","_cell_guid":"d821343d-ccc3-4bf0-af86-9c4733bea43e","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-16T05:51:41.864491Z","iopub.execute_input":"2025-10-16T05:51:41.864767Z","iopub.status.idle":"2025-10-16T05:51:46.599504Z","shell.execute_reply.started":"2025-10-16T05:51:41.864745Z","shell.execute_reply":"2025-10-16T05:51:46.598592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.columns = ['_'.join(col).strip() for col in df.columns.values]\ndf['ImageID'] = df.index\ndf = df.rename(columns={\n    'Label_any': 'any',\n    'Label_epidural': 'epidural',\n    'Label_intraparenchymal': 'intraparenchymal',\n    'Label_intraventricular': 'intraventricular',\n    'Label_subarachnoid': 'subarachnoid',\n    'Label_subdural': 'subdural'\n})\n\ntrain_idx, valid_idx = train_test_split(\n    df['ImageID'], \n    test_size=0.15, \n    random_state=42,\n    stratify=df['any']\n)\n\ntrain_idx, test_idx = train_test_split(\n    train_idx, \n    test_size=0.15, \n    random_state=42,\n    stratify=df.loc[df['ImageID'].isin(train_idx), 'any']\n)\n\ntrain_df = df[df['ImageID'].isin(train_idx)].set_index('ImageID')\nvalid_df = df[df['ImageID'].isin(valid_idx)].set_index('ImageID')\ntest_df  = df[df['ImageID'].isin(test_idx)].set_index('ImageID')\n\nprint(\"Train:\", len(train_df), \"Val:\", len(valid_df), \"Test:\", len(test_df))","metadata":{"_uuid":"7c5b3f9d-5dd4-4e2e-8138-7b29099f253d","_cell_guid":"d143d4dd-8265-4682-9644-d0b8677c5cf6","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df.columns)","metadata":{"_uuid":"cdffe121-efd5-4071-853d-7d8fede44a2f","_cell_guid":"50f2d1c0-92c5-44a9-b7a1-1ad5d3e818ee","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-10-16T06:00:58.298556Z","iopub.execute_input":"2025-10-16T06:00:58.299026Z","iopub.status.idle":"2025-10-16T06:00:58.303186Z","shell.execute_reply.started":"2025-10-16T06:00:58.299002Z","shell.execute_reply":"2025-10-16T06:00:58.302486Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Training InceptionV3 ...\")\ninception_preds = inception_model.fit_and_predict(train_df, valid_df, test_df)\ninception_history = inception_model.history\nprint(\"Training ResNet50 ...\")\nresnet_preds = resnet_model.fit_and_predict(train_df, valid_df, test_df)\nresnet_history = resnet_model.history\nprint(\"Training DenseNet ...\")\ndense_preds = dense_model.fit_and_predict(train_df, valid_df, test_df)\ndense_history = dense_model.history","metadata":{"_uuid":"dbc14ebd-0bf2-4d54-aac5-083cb4753439","_cell_guid":"785af0e7-7318-4923-a9b0-1bebf3a25c90","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-10-16T06:05:51.655167Z","iopub.execute_input":"2025-10-16T06:05:51.655909Z","iopub.status.idle":"2025-10-16T07:48:40.164887Z","shell.execute_reply.started":"2025-10-16T06:05:51.655875Z","shell.execute_reply":"2025-10-16T07:48:40.163187Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\n\ninception_path = \"/kaggle/input/history/inception_history.json\"\nresnet_path = \"/kaggle/input/history/resnet_history.json\"\ndensenet_path = \"/kaggle/input/history/densenet_history.json\"\n\n# Load history\nwith open(inception_path) as f:\n    inception_history = json.load(f)\n\nwith open(resnet_path) as f:\n    resnet_history = json.load(f)\n\nwith open(densenet_path) as f:\n    dense_history = json.load(f)\n","metadata":{"_uuid":"0417bfa7-dc95-45b8-948f-70badac7c8a3","_cell_guid":"ca528f92-9638-45f5-a732-4f40d53b28c7","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-10-16T09:14:27.199728Z","iopub.execute_input":"2025-10-16T09:14:27.200518Z","iopub.status.idle":"2025-10-16T09:14:27.213599Z","shell.execute_reply.started":"2025-10-16T09:14:27.200488Z","shell.execute_reply":"2025-10-16T09:14:27.213Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nhist = inception_history.history\n\nplt.figure(figsize=(10,5))\nplt.subplot(1,2,1)\nplt.plot(hist['loss'], label='Train Loss')\nplt.plot(hist['val_loss'], label='Val Loss')\nplt.title('InceptionV3 - Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.grid()\n\nplt.subplot(1,2,2)\nplt.plot(hist['auc'], label='Train AUC')\nplt.plot(hist['val_auc'], label='Val AUC')\nplt.title('InceptionV3 - AUC')\nplt.xlabel('Epoch')\nplt.ylabel('AUC')\nplt.legend()\nplt.grid()\n\nplt.tight_layout()\nplt.show()\n","metadata":{"_uuid":"b8ac6365-602b-4e5b-a07d-d89e9f323d00","_cell_guid":"1514e616-9fdf-40b4-b94f-5d196b08cfa0","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-10-16T08:03:54.978625Z","iopub.execute_input":"2025-10-16T08:03:54.978924Z","iopub.status.idle":"2025-10-16T08:03:55.339948Z","shell.execute_reply.started":"2025-10-16T08:03:54.978903Z","shell.execute_reply":"2025-10-16T08:03:55.339196Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hist = resnet_history.history\n\nplt.figure(figsize=(10,5))\nplt.subplot(1,2,1)\nplt.plot(hist['loss'], label='Train Loss')\nplt.plot(hist['val_loss'], label='Val Loss')\nplt.title('ResNet50 - Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.grid()\n\nplt.subplot(1,2,2)\nplt.plot(hist['auc'], label='Train AUC')\nplt.plot(hist['val_auc'], label='Val AUC')\nplt.title('ResNet50 - AUC')\nplt.xlabel('Epoch')\nplt.ylabel('AUC')\nplt.legend()\nplt.grid()\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T08:04:15.970694Z","iopub.execute_input":"2025-10-16T08:04:15.971479Z","iopub.status.idle":"2025-10-16T08:04:16.33381Z","shell.execute_reply.started":"2025-10-16T08:04:15.971451Z","shell.execute_reply":"2025-10-16T08:04:16.333143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hist = dense_history.history\n\nplt.figure(figsize=(10,5))\nplt.subplot(1,2,1)\nplt.plot(hist['loss'], label='Train Loss')\nplt.plot(hist['val_loss'], label='Val Loss')\nplt.title('DenseNet121 - Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.grid()\n\nplt.subplot(1,2,2)\nplt.plot(hist['auc'], label='Train AUC')\nplt.plot(hist['val_auc'], label='Val AUC')\nplt.title('DenseNet121 - AUC')\nplt.xlabel('Epoch')\nplt.ylabel('AUC')\nplt.legend()\nplt.grid()\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T08:04:28.022446Z","iopub.execute_input":"2025-10-16T08:04:28.023144Z","iopub.status.idle":"2025-10-16T08:04:28.389666Z","shell.execute_reply.started":"2025-10-16T08:04:28.02312Z","shell.execute_reply":"2025-10-16T08:04:28.388817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\n\nplt.subplot(1,2,1)\nplt.plot(inception_history.history['val_auc'], label='InceptionV3')\nplt.plot(resnet_history.history['val_auc'], label='ResNet50')\nplt.plot(dense_history.history['val_auc'], label='DenseNet121')\nplt.title('Validation AUC Comparison')\nplt.xlabel('Epoch')\nplt.ylabel('AUC')\nplt.legend()\nplt.grid()\n\n# So sánh Validation Loss\nplt.subplot(1,2,2)\nplt.plot(inception_history.history['val_loss'], label='InceptionV3')\nplt.plot(resnet_history.history['val_loss'], label='ResNet50')\nplt.plot(dense_history.history['val_loss'], label='DenseNet121')\nplt.title('Validation Loss Comparison')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.grid()\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T08:06:56.004332Z","iopub.execute_input":"2025-10-16T08:06:56.004832Z","iopub.status.idle":"2025-10-16T08:06:56.370178Z","shell.execute_reply.started":"2025-10-16T08:06:56.004807Z","shell.execute_reply":"2025-10-16T08:06:56.36946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def compare_last_epoch_metrics(histories, model_names):\n    data = []\n    for hist, name in zip(histories, model_names):\n        hist_dict = hist.history\n        last_epoch = len(hist_dict['loss']) - 1\n        row = {'Model': name}\n        for key, values in hist_dict.items():\n            row[key] = values[last_epoch]\n        data.append(row)\n\n    df = pd.DataFrame(data)\n    df = df.set_index('Model')\n    print(df.round(4))  \n    return df\ndf_compare = compare_last_epoch_metrics(\n    [inception_history, resnet_history, dense_history],\n    [\"InceptionV3\", \"ResNet50\", \"DenseNet121\"]\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T08:48:42.68064Z","iopub.execute_input":"2025-10-16T08:48:42.680955Z","iopub.status.idle":"2025-10-16T08:48:42.706978Z","shell.execute_reply.started":"2025-10-16T08:48:42.680933Z","shell.execute_reply":"2025-10-16T08:48:42.706316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"threshold = 0.5\ninception_labels = (inception_preds > threshold).astype(int)\nresnet_labels = (resnet_preds > threshold).astype(int)\ndense_labels = (dense_preds > threshold).astype(int)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T08:50:09.055946Z","iopub.execute_input":"2025-10-16T08:50:09.056207Z","iopub.status.idle":"2025-10-16T08:50:09.060884Z","shell.execute_reply.started":"2025-10-16T08:50:09.056189Z","shell.execute_reply":"2025-10-16T08:50:09.06027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_confusion_per_class(y_true, y_pred, class_names, model_name):\n    n_classes = len(class_names)\n    fig, axes = plt.subplots(2, 3, figsize=(15, 10))  # 6 class → 2 hàng 3 cột\n    fig.suptitle(f'Confusion Matrix per Class - {model_name}', fontsize=16)\n    \n    for i, ax in enumerate(axes.flat):\n        cm = confusion_matrix(y_true[:, i], y_pred[:, i])\n        sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', cbar=False, ax=ax)\n        ax.set_title(class_names[i])\n        ax.set_xlabel('Predicted')\n        ax.set_ylabel('Actual')\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T09:14:52.994384Z","iopub.execute_input":"2025-10-16T09:14:52.994999Z","iopub.status.idle":"2025-10-16T09:14:53.000556Z","shell.execute_reply.started":"2025-10-16T09:14:52.994975Z","shell.execute_reply":"2025-10-16T09:14:52.999918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import precision_score, recall_score, f1_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T08:59:17.347091Z","iopub.execute_input":"2025-10-16T08:59:17.347367Z","iopub.status.idle":"2025-10-16T08:59:17.351571Z","shell.execute_reply.started":"2025-10-16T08:59:17.347347Z","shell.execute_reply":"2025-10-16T08:59:17.35068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_class_report(y_true, y_pred, model_name, class_names):\n    precisions = []\n    recalls = []\n    f1s = []\n\n    for i in range(len(class_names)):\n        p = precision_score(y_true[:, i], y_pred[:, i], zero_division=0)\n        r = recall_score(y_true[:, i], y_pred[:, i], zero_division=0)\n        f1 = f1_score(y_true[:, i], y_pred[:, i], zero_division=0)\n        precisions.append(p)\n        recalls.append(r)\n        f1s.append(f1)\n\n    df = pd.DataFrame({\n        \"Class\": class_names,\n        f\"{model_name}_Precision\": precisions,\n        f\"{model_name}_Recall\": recalls,\n        f\"{model_name}_F1\": f1s\n    })\n    return df\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T08:59:21.417688Z","iopub.execute_input":"2025-10-16T08:59:21.418055Z","iopub.status.idle":"2025-10-16T08:59:21.423675Z","shell.execute_reply.started":"2025-10-16T08:59:21.418029Z","shell.execute_reply":"2025-10-16T08:59:21.423034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndf_incep = get_class_report(y_test, inception_labels, \"InceptionV3\", class_names)\ndf_resnet = get_class_report(y_test, resnet_labels, \"ResNet50\", class_names)\ndf_dense = get_class_report(y_test, dense_labels, \"DenseNet121\", class_names)\n\ndf_metrics = df_incep.merge(df_resnet, on=\"Class\").merge(df_dense, on=\"Class\")\ndf_metrics = df_metrics.set_index(\"Class\")\nprint(df_metrics.round(4))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T08:59:24.192773Z","iopub.execute_input":"2025-10-16T08:59:24.193472Z","iopub.status.idle":"2025-10-16T08:59:24.373546Z","shell.execute_reply.started":"2025-10-16T08:59:24.193446Z","shell.execute_reply":"2025-10-16T08:59:24.372812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_metrics[[ \n    \"InceptionV3_F1\", \n    \"ResNet50_F1\", \n    \"DenseNet121_F1\"\n]].plot(kind='bar', figsize=(12,6))\nplt.title(\"F1-score per class for 3 models\")\nplt.ylabel(\"F1-score\")\nplt.grid(axis='y')\nplt.xticks(rotation=30)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T09:00:06.242252Z","iopub.execute_input":"2025-10-16T09:00:06.242728Z","iopub.status.idle":"2025-10-16T09:00:06.500652Z","shell.execute_reply.started":"2025-10-16T09:00:06.242702Z","shell.execute_reply":"2025-10-16T09:00:06.499903Z"}},"outputs":[],"execution_count":null}]}