{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":99552,"databundleVersionId":13694723,"sourceType":"competition"},{"sourceId":567301,"sourceType":"modelInstanceVersion","modelInstanceId":427458,"modelId":444465}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 一、环境配置","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pydicom\nimport os\nimport json\nfrom pathlib import Path\nfrom skimage import exposure\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import roc_auc_score\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nimport nibabel as nib\nfrom scipy import ndimage","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T11:27:19.68992Z","iopub.execute_input":"2025-09-11T11:27:19.69012Z","iopub.status.idle":"2025-09-11T11:27:34.684301Z","shell.execute_reply.started":"2025-09-11T11:27:19.690103Z","shell.execute_reply":"2025-09-11T11:27:34.68355Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 二、数据预处理","metadata":{}},{"cell_type":"code","source":"# 设置随机种子以确保可重复性\nSEED = 42\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)\n\n# 定义文件路径\nBASE_PATH = '/kaggle/input/rsna-intracranial-aneurysm-detection'\nTRAIN_CSV_PATH = f'{BASE_PATH}/train.csv'\nSERIES_PATH = f'{BASE_PATH}/train'\n\n# 加载训练数据\ntrain_df = pd.read_csv(TRAIN_CSV_PATH)\nprint(f\"训练集大小: {train_df.shape}\")\n\n# 检查数据列\nprint(\"数据列:\", train_df.columns.tolist())\nprint(\"前几行数据:\")\nprint(train_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T11:27:34.685957Z","iopub.execute_input":"2025-09-11T11:27:34.686543Z","iopub.status.idle":"2025-09-11T11:27:34.728324Z","shell.execute_reply.started":"2025-09-11T11:27:34.686524Z","shell.execute_reply":"2025-09-11T11:27:34.727623Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 三、图像处理","metadata":{}},{"cell_type":"code","source":"def load_dicom_series(series_path, series_uid):\n    \"\"\"加载DICOM系列并转换为3D体积\"\"\"\n    try:\n        series_dir = Path(series_path) / series_uid\n        if not series_dir.exists():\n            print(f\"警告: 目录 {series_dir} 不存在\")\n            return np.zeros((64, 64, 64)), []\n            \n        dicom_files = list(series_dir.glob(\"*.dcm\"))\n        \n        if not dicom_files:\n            print(f\"警告: 在 {series_dir} 中未找到DICOM文件\")\n            return np.zeros((64, 64, 64)), []\n        \n        # 按切片位置排序\n        slices = []\n        positions = []\n        \n        for file_path in dicom_files:\n            try:\n                dicom = pydicom.dcmread(str(file_path))\n                slices.append(dicom)\n                if hasattr(dicom, 'ImagePositionPatient') and dicom.ImagePositionPatient:\n                    positions.append(float(dicom.ImagePositionPatient[2]))\n                else:\n                    # 如果没有位置信息，使用切片位置\n                    positions.append(float(getattr(dicom, 'SliceLocation', 0)))\n            except Exception as e:\n                print(f\"读取DICOM文件时出错 {file_path}: {e}\")\n                continue\n        \n        # 检查是否有有效的切片\n        if not slices:\n            print(f\"警告: 没有有效的DICOM切片可用于系列 {series_uid}\")\n            return np.zeros((64, 64, 64)), []\n            \n        # 按位置排序\n        if positions and len(set(positions)) > 1:\n            sorted_slices = [s for _, s in sorted(zip(positions, slices), key=lambda x: x[0])]\n        else:\n            sorted_slices = slices\n        \n        # 创建3D体积\n        try:\n            # 确保所有切片具有相同的尺寸\n            first_shape = sorted_slices[0].pixel_array.shape\n            for i, s in enumerate(sorted_slices):\n                if s.pixel_array.shape != first_shape:\n                    print(f\"警告: 切片 {i} 的尺寸不一致\")\n                    # 可以在这里添加调整尺寸的逻辑，或者跳过不一致的切片\n            \n            volume = np.stack([s.pixel_array for s in sorted_slices], axis=-1)\n            \n            # 应用Rescale斜率/截距\n            if hasattr(sorted_slices[0], 'RescaleSlope') and hasattr(sorted_slices[0], 'RescaleIntercept'):\n                slope = sorted_slices[0].RescaleSlope\n                intercept = sorted_slices[0].RescaleIntercept\n                volume = volume * slope + intercept\n            \n            return volume, sorted_slices\n        except Exception as e:\n            print(f\"创建3D体积时出错: {e}\")\n            return np.zeros((64, 64, 64)), sorted_slices\n            \n    except Exception as e:\n        print(f\"加载DICOM系列时发生未知错误: {e}\")\n        return np.zeros((64, 64, 64)), []\n\ndef preprocess_volume(volume, target_shape=(64, 64, 64)):\n    \"\"\"预处理3D体积\"\"\"\n    if volume.size == 0 or np.all(volume == 0):\n        return np.zeros((*target_shape, 1))\n    \n    # 重采样到目标形状\n    try:\n        zoom_factors = [\n            target_shape[0] / volume.shape[0],\n            target_shape[1] / volume.shape[1],\n            target_shape[2] / volume.shape[2]\n        ]\n        \n        volume_resampled = ndimage.zoom(volume, zoom_factors, order=1)\n    except:\n        # 如果重采样失败，使用零填充\n        volume_resampled = np.zeros(target_shape)\n        min_shape = [min(volume_resampled.shape[i], volume.shape[i]) for i in range(3)]\n        for i in range(min_shape[0]):\n            for j in range(min_shape[1]):\n                for k in range(min_shape[2]):\n                    volume_resampled[i, j, k] = volume[i, j, k]\n    \n    # 强度归一化\n    volume_normalized = (volume_resampled - np.mean(volume_resampled)) / (np.std(volume_resampled) + 1e-6)\n    \n    # 添加通道维度\n    volume_normalized = np.expand_dims(volume_normalized, axis=-1)\n    \n    return volume_normalized\n\ndef augment_volume(volume):\n    \"\"\"对3D体积进行数据增强\"\"\"\n    # 随机翻转\n    if np.random.rand() > 0.5:\n        volume = np.flip(volume, axis=0)  # 沿x轴翻转\n    if np.random.rand() > 0.5:\n        volume = np.flip(volume, axis=1)  # 沿y轴翻转\n    if np.random.rand() > 0.5:\n        volume = np.flip(volume, axis=2)  # 沿z轴翻转\n    \n    # 随机旋转 (0, 90, 180, 270度)\n    k = np.random.randint(0, 4)\n    volume = np.rot90(volume, k=k, axes=(0, 1))\n    \n    # 随机亮度调整\n    if np.random.rand() > 0.5:\n        factor = np.random.uniform(0.8, 1.2)\n        volume = volume * factor\n    \n    return volume","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T11:27:34.729056Z","iopub.execute_input":"2025-09-11T11:27:34.729276Z","iopub.status.idle":"2025-09-11T11:27:34.969732Z","shell.execute_reply.started":"2025-09-11T11:27:34.729259Z","shell.execute_reply":"2025-09-11T11:27:34.968914Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 四、数据生成器","metadata":{}},{"cell_type":"code","source":"class AneurysmClassificationGenerator(tf.keras.utils.Sequence):\n    \"\"\"3D动脉瘤分类数据生成器\"\"\"\n    \n    def __init__(self, series_df, base_path, batch_size=4, \n                 target_shape=(64, 64, 64), shuffle=True, use_cache=True, augment=False):\n        self.series_df = series_df.reset_index(drop=True)\n        self.base_path = base_path\n        self.batch_size = batch_size\n        self.target_shape = target_shape\n        self.shuffle = shuffle\n        self.use_cache = use_cache\n        self.augment = augment\n        self.cache = {}  # 用于缓存预处理后的体积\n        self.on_epoch_end()\n    \n    def __len__(self):\n        return int(np.ceil(len(self.series_df) / self.batch_size))\n    \n    def __getitem__(self, index):\n        batch_indices = self.indices[index*self.batch_size:(index+1)*self.batch_size]\n        batch_series = self.series_df.iloc[batch_indices]\n        \n        X_volumes = np.empty((len(batch_series), *self.target_shape, 1))\n        y = np.empty((len(batch_series), 1))\n        \n        for i, idx in enumerate(batch_indices):\n            series_uid = self.series_df.iloc[idx]['SeriesInstanceUID']\n            \n            # 检查是否已缓存\n            if self.use_cache and series_uid in self.cache:\n                processed_volume = self.cache[series_uid]\n            else:\n                # 加载和预处理3D体积\n                volume, _ = load_dicom_series(self.base_path, series_uid)\n                processed_volume = preprocess_volume(volume, self.target_shape)\n                if self.use_cache:\n                    self.cache[series_uid] = processed_volume\n            \n            # 数据增强\n            if self.augment:\n                processed_volume = augment_volume(processed_volume)\n            \n            X_volumes[i] = processed_volume\n            \n            # 获取标签\n            y[i] = self.series_df.iloc[idx].get('Aneurysm Present', 0)\n        \n        return X_volumes, y\n    \n    def on_epoch_end(self):\n        self.indices = np.arange(len(self.series_df))\n        if self.shuffle:\n            np.random.shuffle(self.indices)\n        # 清空缓存，确保每个epoch使用不同的数据增强\n        self.cache.clear()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T11:27:34.970379Z","iopub.execute_input":"2025-09-11T11:27:34.970568Z","iopub.status.idle":"2025-09-11T11:27:34.983961Z","shell.execute_reply.started":"2025-09-11T11:27:34.970542Z","shell.execute_reply":"2025-09-11T11:27:34.983368Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 五、构建模型","metadata":{}},{"cell_type":"code","source":"def create_3d_classification_model(input_shape, num_classes=1):\n    \"\"\"创建3D动脉瘤分类模型\"\"\"\n    # 输入层\n    inputs = layers.Input(shape=input_shape)\n    \n    # 3D特征提取\n    x = layers.Conv3D(16, 3, activation='relu', padding='same')(inputs)\n    x = layers.BatchNormalization()(x)\n    x = layers.MaxPooling3D(2)(x)\n    \n    x = layers.Conv3D(32, 3, activation='relu', padding='same')(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.MaxPooling3D(2)(x)\n    \n    x = layers.Conv3D(64, 3, activation='relu', padding='same')(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.MaxPooling3D(2)(x)\n    \n    x = layers.Conv3D(128, 3, activation='relu', padding='same')(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.GlobalAveragePooling3D()(x)\n    \n    # 添加一些全连接层\n    x = layers.Dense(64, activation='relu')(x)\n    x = layers.Dropout(0.5)(x)\n    x = layers.Dense(32, activation='relu')(x)\n    x = layers.Dropout(0.5)(x)\n    \n    # 输出层\n    outputs = layers.Dense(num_classes, activation='sigmoid')(x)\n    \n    model = models.Model(inputs=inputs, outputs=outputs)\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T11:27:34.984644Z","iopub.execute_input":"2025-09-11T11:27:34.985316Z","iopub.status.idle":"2025-09-11T11:27:34.997816Z","shell.execute_reply.started":"2025-09-11T11:27:34.985293Z","shell.execute_reply":"2025-09-11T11:27:34.997212Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 六、数据准备","metadata":{}},{"cell_type":"code","source":"# 训练设置\nTARGET_SHAPE = (64, 64, 64)  # 缩小尺寸以适应内存\nBATCH_SIZE = 4  # 小批量以适应内存\n\n# 检查是否有足够的GPU内存\ngpus = tf.config.experimental.list_physical_devices('GPU')\nif gpus:\n    try:\n        for gpu in gpus:\n            tf.config.experimental.set_memory_growth(gpu, True)\n    except RuntimeError as e:\n        print(e)\n\n# 创建模型\nmodel = create_3d_classification_model((*TARGET_SHAPE, 1), num_classes=1)\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4),\n    loss='binary_crossentropy',\n    metrics=['accuracy', tf.keras.metrics.AUC(name='auc')]\n)\n\n# 显示模型架构\nmodel.summary()\n\n# 准备数据 - 使用部分数据进行演示\nsample_size = min(50, len(train_df))  # 使用50个样本进行演示\nsample_df = train_df.sample(sample_size, random_state=SEED)\n\n# 检查是否有Aneurysm Present列\nif 'Aneurysm Present' not in sample_df.columns:\n    print(\"警告: 数据集中没有'Aneurysm Present'列，将使用第一列作为标签\")\n    # 假设第一列是标签列\n    label_col = sample_df.columns[0]\nelse:\n    label_col = 'Aneurysm Present'\n\ntrain_series, val_series = train_test_split(\n    sample_df, test_size=0.2, random_state=SEED, \n    stratify=sample_df[label_col] if label_col in sample_df.columns else None\n)\n\nprint(f\"训练样本数: {len(train_series)}, 验证样本数: {len(val_series)}\")\n\ntrain_generator = AneurysmClassificationGenerator(\n    train_series, SERIES_PATH, \n    batch_size=BATCH_SIZE, target_shape=TARGET_SHAPE,\n    augment=True  # 训练时使用数据增强\n)\n\nval_generator = AneurysmClassificationGenerator(\n    val_series, SERIES_PATH, \n    batch_size=BATCH_SIZE, target_shape=TARGET_SHAPE, \n    shuffle=False, augment=False  # 验证时不使用数据增强\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T11:27:34.998534Z","iopub.execute_input":"2025-09-11T11:27:34.998814Z","iopub.status.idle":"2025-09-11T11:27:37.194209Z","shell.execute_reply.started":"2025-09-11T11:27:34.998789Z","shell.execute_reply":"2025-09-11T11:27:37.193466Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 七、模型训练","metadata":{}},{"cell_type":"code","source":"# 修复ModelCheckpoint的文件名问题\ncheckpoint_filepath = 'best_model.weights.h5'  # 使用.weights.h5扩展名\n\n# 定义回调函数\ncallbacks = [\n    ModelCheckpoint(\n        filepath=checkpoint_filepath,\n        monitor='val_auc',\n        save_best_only=True,\n        save_weights_only=True,  # 只保存权重\n        mode='max',\n        verbose=1\n    ),\n    EarlyStopping(\n        monitor='val_auc',\n        patience=5,\n        restore_best_weights=True,\n        mode='max',\n        verbose=1\n    ),\n    ReduceLROnPlateau(\n        monitor='val_auc',\n        factor=0.5,\n        patience=3,\n        min_lr=1e-7,\n        mode='max',\n        verbose=1\n    )\n]\n\n# 训练模型\nprint(\"开始训练模型...\")\nhistory = model.fit(\n    train_generator,\n    epochs=10,  # 减少epoch数量进行演示\n    validation_data=val_generator,\n    callbacks=callbacks,\n    verbose=1\n)\n\n# 加载最佳权重\nmodel.load_weights(checkpoint_filepath)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T11:27:37.196158Z","iopub.execute_input":"2025-09-11T11:27:37.196394Z","iopub.status.idle":"2025-09-11T11:27:53.54188Z","shell.execute_reply.started":"2025-09-11T11:27:37.196376Z","shell.execute_reply":"2025-09-11T11:27:53.54131Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 八、模型评估","metadata":{}},{"cell_type":"code","source":"# 评估模型\nprint(\"评估模型...\")\nval_loss, val_accuracy, val_auc = model.evaluate(val_generator, verbose=1)\nprint(f\"验证集损失: {val_loss:.4f}, 准确率: {val_accuracy:.4f}, AUC: {val_auc:.4f}\")\n\n# 绘制训练历史\nplt.figure(figsize=(12, 4))\nplt.subplot(1, 2, 1)\nplt.plot(history.history['loss'], label='训练损失')\nplt.plot(history.history['val_loss'], label='验证损失')\nplt.title('模型损失')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history['auc'], label='训练AUC')\nplt.plot(history.history['val_auc'], label='验证AUC')\nplt.title('模型AUC')\nplt.xlabel('Epoch')\nplt.ylabel('AUC')\nplt.legend()\n\nplt.tight_layout()\nplt.savefig('training_history.png')\nplt.show()\n\n# 进行预测\nprint(\"进行预测...\")\npredictions = model.predict(val_generator, verbose=1)\nprint(f\"预测结果形状: {predictions.shape}\")\n\n# 计算AUC分数\ntrue_labels = []\nfor i in range(len(val_generator)):\n    _, labels = val_generator[i]\n    true_labels.extend(labels.flatten())\n\ntrue_labels = np.array(true_labels)\nif len(true_labels) > 0:\n    auc_score = roc_auc_score(true_labels, predictions[:len(true_labels)].flatten())\n    print(f\"最终AUC得分: {auc_score:.4f}\")\nelse:\n    print(\"无法计算AUC得分: 没有真实标签数据\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T11:27:53.542611Z","iopub.execute_input":"2025-09-11T11:27:53.542879Z","iopub.status.idle":"2025-09-11T11:27:55.540106Z","shell.execute_reply.started":"2025-09-11T11:27:53.542853Z","shell.execute_reply":"2025-09-11T11:27:55.539411Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 九、文件提交","metadata":{}}]}