{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":99552,"databundleVersionId":13441085}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 第一部分：环境设置与数据加载","metadata":{}},{"cell_type":"code","source":"# 步骤1: 环境设置与数据加载\n# 安装必要的库（在Kaggle Notebook中通常不需要运行，因为预装好了）\n!pip install pydicom opencv-python scikit-image\n\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport cv2\nfrom skimage import exposure\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.metrics import roc_auc_score\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models, applications\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\n\n# 设置随机种子以确保可重复性\nSEED = 42\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)\n\n# 定义文件路径\nBASE_PATH = '/kaggle/input/rsna-intracranial-aneurysm-detection'\nTRAIN_CSV_PATH = f'{BASE_PATH}/train.csv'\nSERIES_PATH = f'{BASE_PATH}/series'\n\n# 加载训练数据\ntrain_df = pd.read_csv(TRAIN_CSV_PATH)\nprint(f\"训练集大小: {train_df.shape}\")\n\n# 正确定义标签列（14个标签）\nlabel_columns = [\n    'Left Infraclinoid Internal Carotid Artery',\n    'Right Infraclinoid Internal Carotid Artery',\n    'Left Supraclinoid Internal Carotid Artery',\n    'Right Supraclinoid Internal Carotid Artery',\n    'Left Middle Cerebral Artery',\n    'Right Middle Cerebral Artery',\n    'Anterior Communicating Artery',\n    'Left Anterior Cerebral Artery',\n    'Right Anterior Cerebral Artery',\n    'Left Posterior Communicating Artery',\n    'Right Posterior Communicating Artery',\n    'Basilar Tip',\n    'Other Posterior Circulation',\n    'Aneurysm Present'\n]\n\n# 特征列\nfeature_columns = ['PatientAge', 'PatientSex', 'Modality']\n\nprint(\"步骤1完成: 环境设置与数据加载\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T05:10:32.866284Z","iopub.execute_input":"2025-09-10T05:10:32.866599Z","iopub.status.idle":"2025-09-10T05:10:36.004158Z","shell.execute_reply.started":"2025-09-10T05:10:32.866567Z","shell.execute_reply":"2025-09-10T05:10:36.003187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 数据可视化 - 标签分布\nplt.figure(figsize=(15, 8))\nlabel_counts = train_df[label_columns].sum()\nplt.bar(range(len(label_counts)), label_counts.values)\nplt.xticks(range(len(label_counts)), label_counts.index, rotation=45, ha='right')\nplt.title('Distribution of training set labels')#训练集标签分布\nplt.ylabel('Distribution of training set labels')#样本数量\nplt.tight_layout()\nplt.show()\n\n# 数据可视化 - 患者年龄分布\nplt.figure(figsize=(10, 6))\nplt.hist(train_df['PatientAge'], bins=30, edgecolor='black')\nplt.title('Age distribution of patients')#患者年龄分布\nplt.xlabel('age')#年龄\nplt.ylabel('frequency')#频数\nplt.show()\n\n# 数据可视化 - 患者性别分布\nplt.figure(figsize=(8, 6))\ngender_counts = train_df['PatientSex'].value_counts()\nplt.bar(gender_counts.index, gender_counts.values)\nplt.title('Distribution of patients genders')#患者性别分布\nplt.xlabel('Gender')#性别\nplt.ylabel('Frequency')#频数\nplt.xticks(range(len(gender_counts)), gender_counts.index)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T05:10:36.005654Z","iopub.execute_input":"2025-09-10T05:10:36.005889Z","iopub.status.idle":"2025-09-10T05:10:36.590186Z","shell.execute_reply.started":"2025-09-10T05:10:36.005865Z","shell.execute_reply":"2025-09-10T05:10:36.589551Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 第二部分：数据预处理","metadata":{}},{"cell_type":"code","source":"# 步骤2: 数据预处理\n# 编码分类变量\nlabel_encoders = {}\nfor col in ['PatientSex', 'Modality']:\n    le = LabelEncoder()\n    train_df[col] = le.fit_transform(train_df[col].astype(str))\n    label_encoders[col] = le\n    print(f\"{col}编码映射: {dict(zip(le.classes_, le.transform(le.classes_)))}\")\n\n# 标准化数值特征\nscaler = StandardScaler()\ntrain_df['PatientAge'] = scaler.fit_transform(train_df[['PatientAge']])\n\n# 计算类别权重\nclass_weights = {}\nfor i, col in enumerate(label_columns):\n    positive_count = train_df[col].sum()\n    negative_count = len(train_df) - positive_count\n    class_weights[i] = negative_count / (positive_count + 1e-6)  # 防止除零错误\n\nprint(\"\\n类别权重:\", class_weights)\n\n# 数据可视化 - 患者性别分布（编码后）\nplt.figure(figsize=(8, 6))\ngender_counts = train_df['PatientSex'].value_counts()\nplt.bar(range(len(gender_counts)), gender_counts.values)\nplt.title('Distribution of patients genders')#患者性别分布\nplt.xlabel('Gender')#性别\nplt.ylabel('Frequency')#频数\nplt.xticks(range(len(gender_counts)), [f'{idx} ({label_encoders[\"PatientSex\"].classes_[idx]})' for idx in gender_counts.index])\nplt.show()\n\nprint(\"步骤2完成: 数据预处理\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T05:10:36.590918Z","iopub.execute_input":"2025-09-10T05:10:36.591155Z","iopub.status.idle":"2025-09-10T05:10:36.724574Z","shell.execute_reply.started":"2025-09-10T05:10:36.591129Z","shell.execute_reply":"2025-09-10T05:10:36.723948Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 第三部分：图像处理函数","metadata":{}},{"cell_type":"code","source":"# 步骤3: 图像处理函数\n# DICOM图像处理函数\ndef read_dicom_file(path):\n    \"\"\"读取DICOM文件并提取像素数据和元信息\"\"\"\n    try:\n        dicom = pydicom.dcmread(path)\n        image = dicom.pixel_array\n        \n        # 应用模态特定的预处理\n        if hasattr(dicom, 'RescaleSlope') and hasattr(dicom, 'RescaleIntercept'):\n            image = image * dicom.RescaleSlope + dicom.RescaleIntercept\n        \n        return image, dicom\n    except:\n        # 如果无法读取DICOM文件，返回零数组\n        return np.zeros((512, 512)), None\n\ndef preprocess_medical_image(image, target_size=(256, 256)):\n    \"\"\"\n    预处理医学图像\n    \"\"\"\n    # 调整大小\n    if image.shape != target_size:\n        image = cv2.resize(image, target_size)\n    \n    # 对比度限制自适应直方图均衡化（CLAHE）\n    image = exposure.equalize_adapthist(image, clip_limit=0.03)\n    \n    # 标准化\n    image = (image - np.mean(image)) / (np.std(image) + 1e-6)\n    \n    # 将单通道图像复制为三通道（适应EfficientNet）\n    image = np.stack([image, image, image], axis=-1)\n    \n    return image\n\nprint(\"步骤3完成: 图像处理函数定义\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T05:10:36.725974Z","iopub.execute_input":"2025-09-10T05:10:36.726173Z","iopub.status.idle":"2025-09-10T05:10:36.73233Z","shell.execute_reply.started":"2025-09-10T05:10:36.726158Z","shell.execute_reply":"2025-09-10T05:10:36.731765Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 可视化一些样本图像\ndef visualize_sample_images(df, num_samples=5):\n    \"\"\"可视化样本图像\"\"\"\n    fig, axes = plt.subplots(1, num_samples, figsize=(15, 5))\n    \n    for i in range(num_samples):\n        sample_row = df.iloc[i]\n        image_path = f\"{SERIES_PATH}/{sample_row['SeriesInstanceUID']}.dcm\"\n        \n        image, _ = read_dicom_file(image_path)\n        processed_image = preprocess_medical_image(image)\n        \n        axes[i].imshow(processed_image[:, :, 0], cmap='gray')\n        axes[i].set_title(f\"样本 {i+1}\")\n        axes[i].axis('off')\n    \n    plt.tight_layout()\n    plt.show()\n\n# 可视化一些样本\nprint(\"随机样本图像可视化:\")\nvisualize_sample_images(train_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T05:10:36.73304Z","iopub.execute_input":"2025-09-10T05:10:36.733265Z","iopub.status.idle":"2025-09-10T05:10:37.279995Z","shell.execute_reply.started":"2025-09-10T05:10:36.733244Z","shell.execute_reply":"2025-09-10T05:10:37.27935Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 第四部分：数据生成器","metadata":{}},{"cell_type":"code","source":"# 步骤4: 数据生成器\n# 创建改进的数据生成器\nclass MedicalDataGenerator(tf.keras.utils.Sequence):\n    \"\"\"自定义医学数据生成器\"\"\"\n    \n    def __init__(self, df, base_path, batch_size=32, target_size=(256, 256), shuffle=True, **kwargs):\n        # 调用父类初始化方法\n        super().__init__(**kwargs)\n        \n        self.df = df.reset_index(drop=True)\n        self.base_path = base_path\n        self.batch_size = batch_size\n        self.target_size = target_size\n        self.shuffle = shuffle\n        self.on_epoch_end()\n    \n    def __len__(self):\n        return int(np.ceil(len(self.df) / self.batch_size))\n    \n    def __getitem__(self, index):\n        batch_indices = self.indices[index*self.batch_size:(index+1)*self.batch_size]\n        batch_df = self.df.iloc[batch_indices]\n        \n        X_images = np.empty((len(batch_df), *self.target_size, 3))  # 改为3通道\n        X_features = np.empty((len(batch_df), len(feature_columns)))\n        y = np.empty((len(batch_df), len(label_columns)))\n        \n        for i, idx in enumerate(batch_indices):\n            row = self.df.iloc[idx]\n            \n            # 加载和预处理图像\n            image_path = f\"{self.base_path}/{row['SeriesInstanceUID']}.dcm\"\n            image, _ = read_dicom_file(image_path)\n            processed_image = preprocess_medical_image(image, self.target_size)\n            X_images[i] = processed_image\n            \n            # 提取特征 (已经预处理过)\n            X_features[i, 0] = row['PatientAge']  # 已经标准化\n            X_features[i, 1] = row['PatientSex']  # 已经编码\n            X_features[i, 2] = row['Modality']    # 已经编码\n            \n            # 提取标签\n            y[i] = row[label_columns].values.astype(np.float32)\n        \n        # 返回格式改为字典形式，符合Keras多输入模型的要求\n        return {'image_input': X_images, 'feature_input': X_features}, y\n    \n    def on_epoch_end(self):\n        self.indices = np.arange(len(self.df))\n        if self.shuffle:\n            np.random.shuffle(self.indices)\n\nprint(\"步骤4完成: 数据生成器定义\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T05:10:37.280806Z","iopub.execute_input":"2025-09-10T05:10:37.281056Z","iopub.status.idle":"2025-09-10T05:10:37.290793Z","shell.execute_reply.started":"2025-09-10T05:10:37.281037Z","shell.execute_reply":"2025-09-10T05:10:37.29Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 第五部分：模型构建","metadata":{}},{"cell_type":"code","source":"# 步骤5: 模型构建\n# 模型架构\ndef create_multi_input_model(image_shape=(256, 256, 3), num_features=3, num_classes=14):\n    \"\"\"\n    创建多输入模型，同时处理图像和特征\n    \"\"\"\n    # 图像输入分支\n    image_input = tf.keras.Input(shape=image_shape, name='image_input')\n    \n    # 使用EfficientNetB0作为基础模型\n    base_model = applications.EfficientNetB0(\n        weights='imagenet',\n        include_top=False,\n        input_shape=image_shape\n    )\n    \n    # 冻结基础层\n    base_model.trainable = False\n    \n    x = base_model(image_input)\n    x = layers.GlobalAveragePooling2D()(x)\n    x = layers.Dropout(0.5)(x)\n    image_features = layers.Dense(128, activation='relu')(x)\n    \n    # 特征输入分支\n    feature_input = tf.keras.Input(shape=(num_features,), name='feature_input')\n    feature_branch = layers.Dense(16, activation='relu')(feature_input)\n    feature_branch = layers.Dropout(0.3)(feature_branch)\n    \n    # 合并分支\n    combined = layers.Concatenate()([image_features, feature_branch])\n    combined = layers.Dense(64, activation='relu')(combined)\n    combined = layers.Dropout(0.3)(combined)\n    \n    # 输出层\n    outputs = layers.Dense(num_classes, activation='sigmoid')(combined)\n    \n    # 创建模型\n    model = tf.keras.Model(inputs=[image_input, feature_input], outputs=outputs)\n    \n    return model\n\n# 创建模型\nmodel = create_multi_input_model()\nmodel.summary()\n\nprint(\"步骤5完成: 模型构建\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T05:10:37.291583Z","iopub.execute_input":"2025-09-10T05:10:37.291826Z","iopub.status.idle":"2025-09-10T05:10:41.435861Z","shell.execute_reply.started":"2025-09-10T05:10:37.29181Z","shell.execute_reply":"2025-09-10T05:10:41.435262Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 第六部分：模型编译与数据准备","metadata":{}},{"cell_type":"code","source":"# 步骤6: 模型编译与数据准备\n# 自定义加权损失函数\ndef weighted_binary_crossentropy(class_weights):\n    weights = tf.constant(list(class_weights.values()), dtype=tf.float32)\n    \n    def loss_function(y_true, y_pred):\n        # 计算基础损失\n        bce = tf.keras.losses.binary_crossentropy(y_true, y_pred)\n        \n        # 应用类别权重\n        weight_vector = tf.reduce_sum(weights * y_true, axis=1)\n        weighted_bce = bce * weight_vector\n        \n        return tf.reduce_mean(weighted_bce)\n    \n    return loss_function\n\n# 编译模型\noptimizer = tf.keras.optimizers.Adam(learning_rate=0.001)\nmodel.compile(\n    optimizer=optimizer,\n    loss=weighted_binary_crossentropy(class_weights),\n    metrics=['accuracy', 'AUC']\n)\n\n# 划分训练集和验证集\ntrain_data, val_data = train_test_split(\n    train_df, \n    test_size=0.2, \n    random_state=SEED,\n    stratify=train_df['Aneurysm Present']\n)\n\n# 创建数据生成器\ntrain_generator = MedicalDataGenerator(train_data, SERIES_PATH, batch_size=16)\nval_generator = MedicalDataGenerator(val_data, SERIES_PATH, batch_size=16, shuffle=False)\n\n# 设置回调函数\ncallbacks = [\n    ModelCheckpoint(\n        'best_model.h5',\n        monitor='val_auc',\n        save_best_only=True,\n        mode='max',\n        verbose=1\n    ),\n    EarlyStopping(\n        monitor='val_auc',\n        patience=5,\n        restore_best_weights=True,\n        mode='max',\n        verbose=1\n    ),\n    ReduceLROnPlateau(\n        monitor='val_loss',\n        factor=0.5,\n        patience=3,\n        min_lr=1e-7,\n        verbose=1\n    )\n]\n\nprint(\"步骤6完成: 模型编译与数据准备\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T05:10:41.436572Z","iopub.execute_input":"2025-09-10T05:10:41.436841Z","iopub.status.idle":"2025-09-10T05:10:41.464188Z","shell.execute_reply.started":"2025-09-10T05:10:41.436823Z","shell.execute_reply":"2025-09-10T05:10:41.463582Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 第七部分：模型训练","metadata":{}},{"cell_type":"code","source":"# 步骤7: 模型训练\n# 训练模型\nhistory = model.fit(\n    train_generator,\n    epochs=10,\n    validation_data=val_generator,\n    callbacks=callbacks,\n    verbose=1\n)\n\nprint(\"步骤7完成: 模型训练\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T05:10:41.464957Z","iopub.execute_input":"2025-09-10T05:10:41.465215Z","iopub.status.idle":"2025-09-10T05:19:41.943679Z","shell.execute_reply.started":"2025-09-10T05:10:41.465194Z","shell.execute_reply":"2025-09-10T05:19:41.942804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 绘制训练历史 - Loss和AUC曲线\nplt.figure(figsize=(12, 5))\n\n# 绘制损失曲线\nplt.subplot(1, 2, 1)\nplt.plot(history.history['loss'], label='training loss')#训练损失\nplt.plot(history.history['val_loss'], label='Validation loss')#验证损失\nplt.title('Model loss')#模型损失\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\n\n# 绘制AUC曲线\nplt.subplot(1, 2, 2)\nplt.plot(history.history['AUC'], label='Training AUC')#训练\nplt.plot(history.history['val_AUC'], label='Verify AUC')#验证\nplt.title('Model AUC')#模型\nplt.xlabel('Epoch')\nplt.ylabel('AUC')\nplt.legend()\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T05:19:41.946067Z","iopub.execute_input":"2025-09-10T05:19:41.946282Z","iopub.status.idle":"2025-09-10T05:19:42.29489Z","shell.execute_reply.started":"2025-09-10T05:19:41.946264Z","shell.execute_reply":"2025-09-10T05:19:42.294326Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 第八部分：模型评估","metadata":{}},{"cell_type":"code","source":"# 步骤8: 模型评估\n# 评估模型\ndef evaluate_model(model, generator):\n    # 获取所有预测和真实标签\n    all_y_true = []\n    all_y_pred = []\n    \n    for i in range(len(generator)):\n        X, y_true = generator[i]\n        y_pred = model.predict(X, verbose=0)\n        \n        all_y_true.append(y_true)\n        all_y_pred.append(y_pred)\n    \n    y_true = np.concatenate(all_y_true, axis=0)\n    y_pred = np.concatenate(all_y_pred, axis=0)\n    \n    # 计算每个标签的AUC\n    auc_scores = []\n    for i, col in enumerate(label_columns):\n        try:\n            auc = roc_auc_score(y_true[:, i], y_pred[:, i])\n            auc_scores.append(auc)\n            print(f\"{col}: AUC = {auc:.4f}\")\n        except:\n            auc_scores.append(0.5)\n            print(f\"{col}: AUC计算失败，使用默认值0.5\")\n    \n    # 计算加权AUC\n    weights = [1] * 13 + [13]  # 13个部位权重1，存在性权重13\n    weighted_auc = np.average(auc_scores, weights=weights)\n    \n    # 计算最终得分\n    final_score = 0.5 * (auc_scores[-1] + np.mean(auc_scores[:-1]))\n    \n    print(f\"\\n加权AUC: {weighted_auc:.4f}\")\n    print(f\"最终得分: {final_score:.4f}\")\n    \n    return y_true, y_pred, auc_scores, weighted_auc, final_score\n\n# 在验证集上评估模型\nprint(\"模型评估结果:\")\ny_true, y_pred, auc_scores, weighted_auc, final_score = evaluate_model(model, val_generator)\n\nprint(\"步骤8完成: 模型评估\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T05:19:42.295564Z","iopub.execute_input":"2025-09-10T05:19:42.29579Z","iopub.status.idle":"2025-09-10T05:20:11.626058Z","shell.execute_reply.started":"2025-09-10T05:19:42.295774Z","shell.execute_reply":"2025-09-10T05:20:11.625249Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 第九部分：提交文件生成","metadata":{}},{"cell_type":"code","source":"# 步骤9: 提交文件生成\n# 创建提交函数\ndef create_submission(model, test_df, base_path, label_encoders, scaler):\n    \"\"\"\n    创建竞赛提交文件\n    \"\"\"\n    # 预处理测试数据（与训练数据相同的方式）\n    test_df_processed = test_df.copy()\n    \n    # 编码分类变量\n    for col, le in label_encoders.items():\n        # 处理未知类别\n        test_df_processed[col] = test_df_processed[col].apply(\n            lambda x: le.transform([x])[0] if x in le.classes_ else 0\n        )\n    \n    # 标准化数值特征\n    if 'PatientAge' in test_df_processed.columns:\n        test_df_processed['PatientAge'] = scaler.transform(test_df_processed[['PatientAge']])\n    \n    # 创建测试数据生成器\n    test_generator = MedicalDataGenerator(\n        test_df_processed, \n        base_path, \n        batch_size=16, \n        shuffle=False\n    )\n    \n    # 生成预测\n    predictions = model.predict(test_generator, verbose=1)\n    \n    # 创建提交DataFrame\n    submission_df = pd.DataFrame(predictions, columns=label_columns)\n    submission_df.insert(0, 'ID', test_df['SeriesInstanceUID'])\n    \n    # 保存为CSV\n    submission_df.to_csv('submission.csv', index=False)\n    \n    return submission_df\n\n# 注意：在实际使用中，你需要加载测试集数据\n# test_df = pd.read_csv('/kaggle/input/rsna-intracranial-aneurysm-detection/test.csv')\n# submission = create_submission(model, test_df, SERIES_PATH, label_encoders, scaler)\n\nprint(\"步骤9完成: 提交文件生成函数定义\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T05:20:11.62707Z","iopub.execute_input":"2025-09-10T05:20:11.627359Z","iopub.status.idle":"2025-09-10T05:20:11.634167Z","shell.execute_reply.started":"2025-09-10T05:20:11.627333Z","shell.execute_reply":"2025-09-10T05:20:11.633659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}