{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"medica_cleaning_version":"6.0-release","source_notebook":"datacleaning_final_preprocessed_only_v5_auto.ipynb"},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# VinDr-CXR Nodule/Mass 数据清洗发布版 V6\n\n本 Notebook **只读取** `640version1` 已保存的预处理输出，不读取原始\nDICOM，也不读取原始 `train.csv`。它生成一份独立、可追溯、可直接交给\nUltralytics YOLO 的最终数据集。\n\nV6 的正式数据流：\n\n1. 锁定且只接受配置签名为 `b68b716ab287` 的 `640version1` 输出；\n2. 验证 15,000 张源清单、三份 split、融合框和几何缩放契约；\n3. 在导出前解码全部 PNG；可变宽高属于正常输入，实际尺寸与 manifest 不一致时停止并报告，不自动删除；\n4. 同时计算文件 SHA-256 与解码像素 SHA-256；\n5. 对像素完全相同且最终标签相同的副本只生成报告，不自动删除；\n6. 对像素相同但标签冲突的整组样本隔离，并严格阻断正式训练；\n7. 从原始坐标融合框重新生成 YOLO 标签，并把序列化标签投影回实际 PNG；\n8. 同步生成 manifest、split plan、图片、标签、数据指纹与审计报告；\n9. 对导出结果再做一次全量解码、文件哈希、像素哈希和泄漏复查；\n10. 只有全部硬性检查通过，才把训练闸门写成 `TRAINING_READY=True`。\n\n自动处理只发生在 V6 的新输出目录中，**绝不修改或删除 640version1\n源文件**。低对比度、单医生框和感知哈希疑似近重复只进入复核报告，\n不会被武断删除。磁盘 PNG 不要求固定为 640×640；YOLO 在训练时通过\n`imgsz=640` 等比例缩放并补边。\n\n文献依据：\n\n- Nguyen et al., *VinDr-CXR: An open dataset of chest X-rays with\n  radiologist's annotations*, Scientific Data (2022),\n  DOI: `10.1038/s41597-022-01498-w`.\n- Kapoor & Narayanan, *Leakage and the reproducibility crisis in\n  machine-learning-based science*, Patterns (2023),\n  DOI: `10.1016/j.patter.2023.100804`.\n","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. 导入：仅 CPU 数据处理库\nimport hashlib\nimport json\nimport math\nimport os\nimport platform\nimport random\nimport shutil\nfrom concurrent.futures import ThreadPoolExecutor\nfrom datetime import datetime, timezone\nfrom pathlib import Path\nfrom time import perf_counter\n\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport yaml\nfrom PIL import Image, ImageDraw\n\ntry:\n    from tqdm.auto import tqdm\nexcept ImportError:\n    def tqdm(iterable, **_):\n        return iterable\n\ntry:\n    from IPython.display import display\nexcept ImportError:\n    display = print\n\nSEED = 2026\nrandom.seed(SEED)\nnp.random.seed(SEED)\n\nprint(\"Python:\", platform.python_version())\nprint(\"CPU cores:\", os.cpu_count())\nprint(\"GPU used by this notebook: False\")\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. 发布配置：默认值已锁定到本次验证的 640version1 输出\nINPUT_ROOT = Path(os.environ.get(\"VINDR_INPUT_ROOT\", \"/kaggle/input\"))\nOUTPUT_BASE_ROOT = Path(\n    os.environ.get(\n        \"VINDR_OUTPUT_ROOT\",\n        \"/kaggle/working/vindr_nodule_mass_cleaning_final_v6\",\n    )\n)\n\nPREPROCESSED_DATASET_ROOT = Path(\n    os.environ.get(\n        \"VINDR_PREPROCESSED_ROOT\",\n        (\n            \"/kaggle/input/notebooks/hilarylee33/640version1/\"\n            \"vindr_nodule_mass_training/datasets/b68b716ab287\"\n        ),\n    )\n)\nEXPECTED_SOURCE_NOTEBOOK_ROOT = Path(\n    os.environ.get(\n        \"VINDR_640VERSION1_ROOT\",\n        \"/kaggle/input/notebooks/hilarylee33/640version1\",\n    )\n)\n\nEXPECTED_SOURCE_OUTPUT_DIRNAME = \"vindr_nodule_mass_training\"\nEXPECTED_SOURCE_NOTEBOOK_HINT = \"640version1\"\nEXPECTED_SOURCE_DATASET_SIGNATURE = \"b68b716ab287\"\nREQUIRE_NOTEBOOK_NAME_HINT = True\n\n# 这四项把 V6 锁定到本次已经审计过的完整源版本，防止误挂子集或旧输出。\nEXPECTED_SOURCE_IMAGE_COUNT = 15000\nEXPECTED_SOURCE_SPLIT_COUNTS = {\n    \"train\": 12000,\n    \"val\": 1500,\n    \"test\": 1500,\n}\nEXPECTED_SOURCE_POSITIVE_IMAGE_COUNT = 826\nEXPECTED_SOURCE_FUSED_BOX_COUNT = 1803\n\nTARGET_CLASS = \"Nodule/Mass\"\nYOLO_CLASS_ID = 0\nEXPECTED_TRAIN_RADIOLOGISTS_PER_IMAGE = 3\n\n# 几何清洗。框完全消失或过小会被隔离，阳性图绝不会被伪装成阴性。\nMIN_VISIBLE_FRACTION = 0.50\nMIN_BOX_SIDE_ORIGINAL_PX = 2.0\nMIN_BOX_SIDE_OUTPUT_PX = 1.0\nALGEBRAIC_GEOMETRY_TOLERANCE_PX = 1e-6\nLABEL_SERIALIZATION_DECIMALS = 8\nSERIALIZED_GEOMETRY_TOLERANCE_OUTPUT_PX = 1e-4\nMAX_AMBIGUOUS_POSITIVE_FRACTION = 0.05\n\n# 可安全自动隔离的源图片问题及其最大允许比例。\nAUTO_EXCLUDE_UNREADABLE_OR_INVALID_IMAGES = True\nMAX_AUTOMATIC_IMAGE_EXCLUSION_FRACTION = 0.005\nMAX_AUTOMATIC_POSITIVE_EXCLUSION_FRACTION = 0.01\n\n# copy：正式保存/共享；symlink：仅当前会话；none：只写报告。\nIMAGE_EXPORT_MODE = \"copy\"\n\n# 正式训练必须 full。sample 只能开发调试。\nAUDIT_MODE = \"full\"\nSAMPLE_IMAGES_PER_SPLIT = 100\nAUDIT_WORKERS = min(8, max(1, os.cpu_count() or 1))\nREQUIRE_GRAYSCALE = True\nCOMPUTE_SHA256 = True\nVERIFY_EXISTING_COPY_HASH = True\nLOW_CONTRAST_STD_REVIEW_THRESHOLD = 3.0\nEXTREME_SATURATION_REVIEW_FRACTION = 0.98\nREPORT_PERCEPTUAL_HASH_COLLISIONS = True\n\nDISK_SPACE_SAFETY_FACTOR = 1.10\nDISK_SPACE_RESERVE_GB = 0.50\n\nFAIL_ON_NOT_TRAINING_READY = True\nPREVIEW_POSITIVE_IMAGES = 6\n\nVALID_SPLITS = (\"train\", \"val\", \"test\")\nAUTO_DEDUPLICATE_EXACT_CONTENT = False\nDUPLICATE_KEEP_PRIORITY = (\"test\", \"val\", \"train\")\nBLOCK_ON_DUPLICATE_LABEL_CONFLICT = True\n\nCOORDINATE_COLUMNS = [\"x_min\", \"y_min\", \"x_max\", \"y_max\"]\nREQUIRED_MANIFEST_COLUMNS = {\n    \"image_id\", \"split\", \"is_positive\", \"label_count\",\n    \"original_width\", \"original_height\", \"output_width\", \"output_height\",\n}\nREQUIRED_FUSED_COLUMNS = {\"image_id\", *COORDINATE_COLUMNS}\n\nif IMAGE_EXPORT_MODE not in {\"copy\", \"symlink\", \"none\"}:\n    raise ValueError(\"IMAGE_EXPORT_MODE 必须是 copy、symlink 或 none。\")\nif AUDIT_MODE not in {\"full\", \"sample\"}:\n    raise ValueError(\"AUDIT_MODE 必须是 full 或 sample。\")\nif not 0 < MIN_VISIBLE_FRACTION <= 1:\n    raise ValueError(\"MIN_VISIBLE_FRACTION 必须在 (0, 1]。\")\nif MIN_BOX_SIDE_ORIGINAL_PX <= 0 or MIN_BOX_SIDE_OUTPUT_PX <= 0:\n    raise ValueError(\"最小框边长必须大于 0。\")\nif LABEL_SERIALIZATION_DECIMALS < 6:\n    raise ValueError(\"YOLO 标签至少保留 6 位小数。\")\nif SERIALIZED_GEOMETRY_TOLERANCE_OUTPUT_PX <= 0:\n    raise ValueError(\"序列化几何误差阈值必须大于 0。\")\nif not 0 <= MAX_AMBIGUOUS_POSITIVE_FRACTION <= 1:\n    raise ValueError(\"MAX_AMBIGUOUS_POSITIVE_FRACTION 必须在 [0, 1]。\")\nif not 0 <= MAX_AUTOMATIC_IMAGE_EXCLUSION_FRACTION <= 1:\n    raise ValueError(\"自动图片隔离比例阈值必须在 [0, 1]。\")\nif not 0 <= MAX_AUTOMATIC_POSITIVE_EXCLUSION_FRACTION <= 1:\n    raise ValueError(\"自动阳性图片隔离比例阈值必须在 [0, 1]。\")\nif AUDIT_WORKERS <= 0:\n    raise ValueError(\"AUDIT_WORKERS 必须大于 0。\")\nif not isinstance(AUTO_DEDUPLICATE_EXACT_CONTENT, bool):\n    raise ValueError(\"AUTO_DEDUPLICATE_EXACT_CONTENT 必须是布尔值。\")\nif set(DUPLICATE_KEEP_PRIORITY) != set(VALID_SPLITS):\n    raise ValueError(\"DUPLICATE_KEEP_PRIORITY 必须且只能包含三份 split。\")\nif not BLOCK_ON_DUPLICATE_LABEL_CONFLICT:\n    raise ValueError(\"V6 发布版要求标签冲突严格阻断训练。\")\nif not COMPUTE_SHA256:\n    raise ValueError(\"V6 依赖 SHA-256，COMPUTE_SHA256 必须为 True。\")\n\nCLEANING_POLICY = {\n    \"version\": \"6.0-release\",\n    \"target_class\": TARGET_CLASS,\n    \"yolo_class_id\": YOLO_CLASS_ID,\n    \"image_export_mode\": IMAGE_EXPORT_MODE,\n    \"min_visible_fraction\": MIN_VISIBLE_FRACTION,\n    \"min_box_side_original_px\": MIN_BOX_SIDE_ORIGINAL_PX,\n    \"min_box_side_output_px\": MIN_BOX_SIDE_OUTPUT_PX,\n    \"max_ambiguous_positive_fraction\": (\n        MAX_AMBIGUOUS_POSITIVE_FRACTION\n    ),\n    \"label_serialization_decimals\": LABEL_SERIALIZATION_DECIMALS,\n    \"serialized_geometry_tolerance_output_px\": (\n        SERIALIZED_GEOMETRY_TOLERANCE_OUTPUT_PX\n    ),\n    \"auto_exclude_invalid_images\": AUTO_EXCLUDE_UNREADABLE_OR_INVALID_IMAGES,\n    \"max_auto_image_exclusion_fraction\": (\n        MAX_AUTOMATIC_IMAGE_EXCLUSION_FRACTION\n    ),\n    \"max_auto_positive_exclusion_fraction\": (\n        MAX_AUTOMATIC_POSITIVE_EXCLUSION_FRACTION\n    ),\n    \"duplicate_hash\": \"decoded_grayscale_pixels_sha256\",\n    \"auto_deduplicate_exact_content\": (\n        AUTO_DEDUPLICATE_EXACT_CONTENT\n    ),\n    \"duplicate_keep_priority\": list(DUPLICATE_KEEP_PRIORITY),\n    \"block_on_duplicate_label_conflict\": (\n        BLOCK_ON_DUPLICATE_LABEL_CONFLICT\n    ),\n    \"require_grayscale\": REQUIRE_GRAYSCALE,\n    \"audit_mode\": AUDIT_MODE,\n    \"seed\": SEED,\n}\nCLEANING_POLICY_SIGNATURE = hashlib.sha256(\n    json.dumps(\n        CLEANING_POLICY, sort_keys=True, ensure_ascii=False\n    ).encode(\"utf-8\")\n).hexdigest()[:12]\n\nOUTPUT_BASE_ROOT.mkdir(parents=True, exist_ok=True)\n\nprint(\"Input root:\", INPUT_ROOT)\nprint(\"Output base root:\", OUTPUT_BASE_ROOT)\nprint(\"Image export mode:\", IMAGE_EXPORT_MODE)\nprint(\"Audit mode:\", AUDIT_MODE)\nprint(\"Audit workers:\", AUDIT_WORKERS)\nprint(\"Cleaning policy signature:\", CLEANING_POLICY_SIGNATURE)\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. 锁定 640version1 预处理输出\n\n程序同时验证目录结构、配置签名、目标类别和来源报告。输出再按源数据配置签名\n建立独立目录，因此不同预处理版本不会混进同一个清洗结果。\n","metadata":{}},{"cell_type":"code","source":"def source_output_root_for(dataset_root: Path):\n    for parent in [dataset_root, *dataset_root.parents]:\n        if parent.name == EXPECTED_SOURCE_OUTPUT_DIRNAME:\n            return parent\n    return None\n\n\ndef candidate_reason(dataset_root: Path):\n    required_paths = [\n        dataset_root / \"manifest.csv\",\n        dataset_root / \"dataset_config.json\",\n        dataset_root / \"data.yaml\",\n        *[dataset_root / \"images\" / split for split in VALID_SPLITS],\n        *[dataset_root / \"labels\" / split for split in VALID_SPLITS],\n    ]\n    missing = [str(path) for path in required_paths if not path.exists()]\n    if (\n        REQUIRE_NOTEBOOK_NAME_HINT\n        and EXPECTED_SOURCE_NOTEBOOK_HINT.lower() not in str(dataset_root).lower()\n    ):\n        missing.append(\n            f\"path containing notebook hint {EXPECTED_SOURCE_NOTEBOOK_HINT!r}\"\n        )\n    output_root = source_output_root_for(dataset_root)\n    if output_root is None:\n        missing.append(\n            f\"ancestor directory named {EXPECTED_SOURCE_OUTPUT_DIRNAME}\"\n        )\n    else:\n        for report_name in (\n            \"fused_boxes_original_coordinates.csv\",\n            \"split_plan.csv\",\n        ):\n            report_path = output_root / \"reports\" / report_name\n            if not report_path.exists():\n                missing.append(str(report_path))\n    return missing\n\n\nif PREPROCESSED_DATASET_ROOT is not None:\n    source_dataset_root = Path(PREPROCESSED_DATASET_ROOT)\n    if source_dataset_root.name == \"manifest.csv\":\n        source_dataset_root = source_dataset_root.parent\n    missing_source_parts = candidate_reason(source_dataset_root)\n    if missing_source_parts:\n        raise FileNotFoundError(\n            \"指定目录不是完整的 640version1 预处理输出，缺少：\\n- \"\n            + \"\\n- \".join(missing_source_parts)\n        )\nelse:\n    # 仅在明确的 640version1 Notebook Output 内查找，禁止扫描整个 Kaggle Input。\n    if not EXPECTED_SOURCE_NOTEBOOK_ROOT.is_dir():\n        raise FileNotFoundError(\n            \"未找到 640version1 Notebook Output：\"\n            f\"{EXPECTED_SOURCE_NOTEBOOK_ROOT}\"\n        )\n    candidate_roots = []\n    for manifest_path in EXPECTED_SOURCE_NOTEBOOK_ROOT.rglob(\"manifest.csv\"):\n        dataset_root = manifest_path.parent\n        if not candidate_reason(dataset_root):\n            candidate_roots.append(dataset_root)\n    candidate_roots = sorted(set(candidate_roots), key=lambda path: str(path))\n    if not candidate_roots:\n        raise FileNotFoundError(\n            \"没有找到完整的 640version1 预处理输出。请先对 640version1 Run All、\"\n            \"Save Version，再把该 Notebook Output 添加为当前 Notebook 的 Input。\"\n        )\n    if len(candidate_roots) > 1:\n        raise RuntimeError(\n            \"发现多个合格输出，为防止误读没有自动选择。\"\n            \"请填写 PREPROCESSED_DATASET_ROOT：\\n- \"\n            + \"\\n- \".join(str(path) for path in candidate_roots)\n        )\n    source_dataset_root = candidate_roots[0]\n\nsource_output_root = source_output_root_for(source_dataset_root)\nsource_manifest_path = source_dataset_root / \"manifest.csv\"\nsource_config_path = source_dataset_root / \"dataset_config.json\"\nsource_data_yaml_path = source_dataset_root / \"data.yaml\"\nsource_image_root = source_dataset_root / \"images\"\nsource_label_root = source_dataset_root / \"labels\"\nsource_report_root = source_output_root / \"reports\"\nsource_fused_boxes_path = (\n    source_report_root / \"fused_boxes_original_coordinates.csv\"\n)\nsource_split_plan_path = source_report_root / \"split_plan.csv\"\n\nsource_config = json.loads(source_config_path.read_text(encoding=\"utf-8\"))\nrequired_config_keys = {\"target_class\", \"preprocess_max_side\", \"seed\"}\nmissing_config_keys = required_config_keys - set(source_config)\nif missing_config_keys:\n    raise RuntimeError(\n        f\"dataset_config.json 缺少关键字段：{sorted(missing_config_keys)}\"\n    )\n\nconfig_text = json.dumps(source_config, sort_keys=True, ensure_ascii=False)\ncomputed_signature = hashlib.sha256(config_text.encode(\"utf-8\")).hexdigest()[:12]\nsignature_matches = source_dataset_root.name == computed_signature\nsource_notebook_hint_matches = (\n    EXPECTED_SOURCE_NOTEBOOK_HINT.lower() in str(source_dataset_root).lower()\n)\nif computed_signature != EXPECTED_SOURCE_DATASET_SIGNATURE:\n    raise RuntimeError(\n        \"源配置签名不是本次发布版锁定值：\"\n        f\"{computed_signature} != {EXPECTED_SOURCE_DATASET_SIGNATURE}。\"\n    )\nif not signature_matches:\n    raise RuntimeError(\n        \"预处理配置签名不匹配：目录名为 \"\n        f\"{source_dataset_root.name}，计算结果为 {computed_signature}。\"\n    )\nif REQUIRE_NOTEBOOK_NAME_HINT and not source_notebook_hint_matches:\n    raise RuntimeError(\n        f\"路径不包含 {EXPECTED_SOURCE_NOTEBOOK_HINT!r}，禁止自动读取。\"\n    )\nif source_config.get(\"target_class\") != TARGET_CLASS:\n    raise RuntimeError(\n        \"预处理目标类别不匹配：\"\n        f\"{source_config.get('target_class')!r} != {TARGET_CLASS!r}\"\n    )\nsource_preprocess_max_side = source_config.get(\"preprocess_max_side\")\nif (\n    source_preprocess_max_side is not None\n    and float(source_preprocess_max_side) <= 0\n):\n    raise RuntimeError(\"preprocess_max_side 必须为空或正数。\")\n\n# 源配置与清洗策略共同隔离输出，避免旧版本或不同阈值残留混入。\nRUN_ROOT = (\n    OUTPUT_BASE_ROOT / computed_signature\n    / f\"policy-{CLEANING_POLICY_SIGNATURE}\"\n)\nREPORT_ROOT = RUN_ROOT / \"reports\"\nDATASET_ROOT = RUN_ROOT / \"dataset\"\nNEW_IMAGE_ROOT = DATASET_ROOT / \"images\"\nNEW_LABEL_ROOT = DATASET_ROOT / \"labels\"\nfor directory in (RUN_ROOT, REPORT_ROOT, DATASET_ROOT, NEW_LABEL_ROOT):\n    directory.mkdir(parents=True, exist_ok=True)\nif IMAGE_EXPORT_MODE != \"none\":\n    NEW_IMAGE_ROOT.mkdir(parents=True, exist_ok=True)\n\n# Fail closed：当前运行完成前，旧的成功闸门绝不能继续显示为可训练。\nRUN_STARTED_AT = datetime.now(timezone.utc).isoformat()\n(REPORT_ROOT / \"TRAINING_GATE.txt\").write_text(\n    \"STATUS=RUNNING\\nTRAINING_READY=False\\n\"\n    f\"RUN_STARTED_AT={RUN_STARTED_AT}\\n\",\n    encoding=\"utf-8\",\n)\n(REPORT_ROOT / \"cleaning_acceptance_final.json\").write_text(\n    json.dumps({\n        \"overall_status\": \"RUNNING\",\n        \"training_ready\": False,\n        \"run_started_at\": RUN_STARTED_AT,\n        \"source_dataset_signature\": computed_signature,\n        \"cleaning_policy_signature\": CLEANING_POLICY_SIGNATURE,\n    }, ensure_ascii=False, indent=2),\n    encoding=\"utf-8\",\n)\n\nsource_provenance = {\n    \"input_policy\": \"640version1_preprocessed_output_only\",\n    \"raw_dicom_read\": False,\n    \"raw_train_csv_read\": False,\n    \"source_dataset_root\": str(source_dataset_root),\n    \"source_output_root\": str(source_output_root),\n    \"source_manifest\": str(source_manifest_path),\n    \"source_dataset_config\": str(source_config_path),\n    \"source_data_yaml\": str(source_data_yaml_path),\n    \"source_fused_boxes\": str(source_fused_boxes_path),\n    \"source_split_plan\": str(source_split_plan_path),\n    \"expected_source_notebook_hint\": EXPECTED_SOURCE_NOTEBOOK_HINT,\n    \"dataset_signature\": computed_signature,\n    \"expected_dataset_signature\": EXPECTED_SOURCE_DATASET_SIGNATURE,\n    \"cleaning_policy\": CLEANING_POLICY,\n    \"cleaning_policy_signature\": CLEANING_POLICY_SIGNATURE,\n    \"run_started_at\": RUN_STARTED_AT,\n    \"dataset_signature_verified\": signature_matches,\n    \"source_notebook_hint_verified\": source_notebook_hint_matches,\n    \"source_configuration\": source_config,\n    \"v6_run_root\": str(RUN_ROOT),\n}\n(REPORT_ROOT / \"source_provenance.json\").write_text(\n    json.dumps(source_provenance, ensure_ascii=False, indent=2),\n    encoding=\"utf-8\",\n)\n\nprint(\"SOURCE LOCK PASSED\")\nprint(\"Only preprocessed output will be read:\", source_dataset_root)\nprint(\"Raw DICOM read: False\")\nprint(\"Raw train.csv read: False\")\nprint(\"Verified dataset signature:\", computed_signature)\nprint(\"V6 run root:\", RUN_ROOT)\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4. 验证 manifest、几何变换契约和预处理文件结构\nmanifest = pd.read_csv(source_manifest_path)\nmissing_manifest_columns = REQUIRED_MANIFEST_COLUMNS - set(manifest.columns)\nif missing_manifest_columns:\n    raise ValueError(\n        f\"manifest 缺少 640version1 必需字段：{sorted(missing_manifest_columns)}\"\n    )\n\nmanifest[\"image_id\"] = manifest[\"image_id\"].astype(str).str.strip()\nmanifest[\"split\"] = manifest[\"split\"].astype(str).str.strip().str.lower()\ninvalid_manifest_image_ids = manifest.loc[\n    manifest[\"image_id\"].str.lower().isin([\"\", \"nan\", \"none\", \"null\"])\n].copy()\nnumeric_manifest_columns = [\n    \"is_positive\", \"label_count\",\n    \"original_width\", \"original_height\", \"output_width\", \"output_height\",\n]\nfor column in numeric_manifest_columns:\n    manifest[column] = pd.to_numeric(manifest[column], errors=\"coerce\")\n\nduplicate_manifest_rows = manifest.loc[\n    manifest.duplicated(\"image_id\", keep=False)\n].copy()\ninvalid_split_rows = manifest.loc[\n    ~manifest[\"split\"].isin(VALID_SPLITS)\n].copy()\ninvalid_dimension_rows = manifest.loc[\n    ~np.isfinite(\n        manifest[\n            [\"original_width\", \"original_height\", \"output_width\", \"output_height\"]\n        ]\n    ).all(axis=1)\n    | (manifest[\"original_width\"] <= 0)\n    | (manifest[\"original_height\"] <= 0)\n    | (manifest[\"output_width\"] <= 0)\n    | (manifest[\"output_height\"] <= 0)\n    | (\n        manifest[\n            [\"original_width\", \"original_height\",\n             \"output_width\", \"output_height\"]\n        ]\n        != np.floor(manifest[\n            [\"original_width\", \"original_height\",\n             \"output_width\", \"output_height\"]\n        ])\n    ).any(axis=1)\n].copy()\ninvalid_status_rows = manifest.loc[\n    ~np.isfinite(manifest[[\"is_positive\", \"label_count\"]]).all(axis=1)\n    | ~manifest[\"is_positive\"].isin([0, 1])\n    | (manifest[\"label_count\"] < 0)\n    | (manifest[\"label_count\"] != np.floor(manifest[\"label_count\"]))\n].copy()\n\ninvalid_manifest_image_ids.to_csv(\n    REPORT_ROOT / \"invalid_manifest_image_ids.csv\", index=False\n)\nduplicate_manifest_rows.to_csv(\n    REPORT_ROOT / \"duplicate_manifest_rows.csv\", index=False\n)\ninvalid_split_rows.to_csv(REPORT_ROOT / \"invalid_split_rows.csv\", index=False)\ninvalid_dimension_rows.to_csv(\n    REPORT_ROOT / \"invalid_manifest_dimensions.csv\", index=False\n)\ninvalid_status_rows.to_csv(\n    REPORT_ROOT / \"invalid_manifest_status.csv\", index=False\n)\nif len(invalid_manifest_image_ids):\n    raise RuntimeError(\"预处理 manifest 含空或非法 image_id。\")\nif len(duplicate_manifest_rows):\n    raise RuntimeError(\"预处理 manifest 含重复 image_id，禁止继续。\")\nif len(invalid_split_rows):\n    raise RuntimeError(\"预处理 manifest 含非法 split，禁止继续。\")\nif len(invalid_dimension_rows):\n    raise RuntimeError(\"预处理 manifest 含非法尺寸，禁止继续。\")\nif len(invalid_status_rows):\n    raise RuntimeError(\"预处理 manifest 含非法阳性状态或框数量，禁止继续。\")\nif set(manifest[\"split\"]) != set(VALID_SPLITS):\n    raise RuntimeError(\"manifest 必须同时包含 train、val、test。\")\n\nactual_source_split_counts = {\n    split: int(count)\n    for split, count in manifest.groupby(\"split\").size().items()\n}\nactual_source_positive_count = int(manifest[\"is_positive\"].sum())\nsource_count_contract = {\n    \"expected_image_count\": EXPECTED_SOURCE_IMAGE_COUNT,\n    \"actual_image_count\": int(len(manifest)),\n    \"expected_split_counts\": EXPECTED_SOURCE_SPLIT_COUNTS,\n    \"actual_split_counts\": actual_source_split_counts,\n    \"expected_positive_image_count\": EXPECTED_SOURCE_POSITIVE_IMAGE_COUNT,\n    \"actual_positive_image_count\": actual_source_positive_count,\n}\n(REPORT_ROOT / \"source_count_contract.json\").write_text(\n    json.dumps(source_count_contract, ensure_ascii=False, indent=2),\n    encoding=\"utf-8\",\n)\nsource_count_contract_pass = (\n    len(manifest) == EXPECTED_SOURCE_IMAGE_COUNT\n    and actual_source_split_counts == EXPECTED_SOURCE_SPLIT_COUNTS\n    and actual_source_positive_count\n    == EXPECTED_SOURCE_POSITIVE_IMAGE_COUNT\n)\nif not source_count_contract_pass:\n    raise RuntimeError(\n        \"源数据数量不符合锁定版本；可能挂载了子集、旧输出或错误版本。\"\n    )\n\n\ndef expected_output_size_contract(original_width, original_height):\n    original_width = int(original_width)\n    original_height = int(original_height)\n    longest = max(original_width, original_height)\n    if (\n        source_preprocess_max_side is None\n        or longest <= float(source_preprocess_max_side)\n    ):\n        return original_width, original_height\n    scale = float(source_preprocess_max_side) / longest\n    return (\n        max(1, round(original_width * scale)),\n        max(1, round(original_height * scale)),\n    )\n\n\ngeometry_contract = manifest[\n    [\n        \"image_id\", \"split\", \"original_width\", \"original_height\",\n        \"output_width\", \"output_height\",\n    ]\n].copy()\nexpected_sizes = geometry_contract.apply(\n    lambda row: expected_output_size_contract(\n        row[\"original_width\"], row[\"original_height\"]\n    ),\n    axis=1,\n)\ngeometry_contract[\"expected_output_width\"] = [\n    size[0] for size in expected_sizes\n]\ngeometry_contract[\"expected_output_height\"] = [\n    size[1] for size in expected_sizes\n]\ngeometry_contract[\"dimension_contract_matches\"] = (\n    geometry_contract[\"output_width\"].astype(int)\n    == geometry_contract[\"expected_output_width\"].astype(int)\n) & (\n    geometry_contract[\"output_height\"].astype(int)\n    == geometry_contract[\"expected_output_height\"].astype(int)\n)\ngeometry_contract[\"scale_x\"] = (\n    geometry_contract[\"output_width\"] / geometry_contract[\"original_width\"]\n)\ngeometry_contract[\"scale_y\"] = (\n    geometry_contract[\"output_height\"] / geometry_contract[\"original_height\"]\n)\ngeometry_contract[\"aspect_ratio_relative_error\"] = (\n    (\n        (geometry_contract[\"output_width\"] / geometry_contract[\"output_height\"])\n        / (\n            geometry_contract[\"original_width\"]\n            / geometry_contract[\"original_height\"]\n        )\n    )\n    - 1\n).abs()\ngeometry_contract.to_csv(\n    REPORT_ROOT / \"geometry_manifest_contract.csv\", index=False\n)\ngeometry_manifest_failures = geometry_contract.loc[\n    ~geometry_contract[\"dimension_contract_matches\"]\n].copy()\ngeometry_manifest_failures.to_csv(\n    REPORT_ROOT / \"geometry_manifest_contract_failures.csv\", index=False\n)\ngeometry_manifest_contract_pass = len(geometry_manifest_failures) == 0\nif not geometry_manifest_contract_pass:\n    raise RuntimeError(\n        \"manifest 输出尺寸不符合 640version1 的等比例缩放契约；\"\n        \"可能存在裁剪、补边、路径混用或 manifest 错配。\"\n    )\n\nresolved_records = []\nfor row in manifest.itertuples(index=False):\n    image_path = source_image_root / row.split / f\"{row.image_id}.png\"\n    label_path = source_label_root / row.split / f\"{row.image_id}.txt\"\n    resolved_records.append({\n        \"image_id\": row.image_id,\n        \"split\": row.split,\n        \"source_image_path\": str(image_path),\n        \"source_label_path\": str(label_path),\n        \"source_image_exists\": image_path.is_file(),\n        \"source_label_exists\": label_path.is_file(),\n    })\nresolved_paths = pd.DataFrame(resolved_records)\nmanifest = manifest.merge(\n    resolved_paths,\n    on=[\"image_id\", \"split\"],\n    how=\"left\",\n    validate=\"one_to_one\",\n)\n\n# 明确生成受来源锁约束的图片清单；后续绝不遍历其他 Input 图片目录。\nSOURCE_IMAGES = tuple(\n    Path(path) for path in manifest[\"source_image_path\"].tolist()\n)\nfor image_path in SOURCE_IMAGES:\n    if source_image_root not in image_path.parents:\n        raise RuntimeError(\n            f\"来源锁违规：图片不在 640version1 images 目录内：{image_path}\"\n        )\n\nmissing_source_files = manifest.loc[\n    ~manifest[\"source_image_exists\"] | ~manifest[\"source_label_exists\"]\n].copy()\nmissing_source_files.to_csv(\n    REPORT_ROOT / \"missing_preprocessed_source_files.csv\", index=False\n)\nif len(missing_source_files):\n    raise FileNotFoundError(\n        f\"预处理输出缺少 {len(missing_source_files)} 组 PNG/标签文件。\"\n    )\n\nexpected_image_keys = set(zip(manifest[\"split\"], manifest[\"image_id\"]))\nactual_image_keys = {\n    (path.parent.name, path.stem)\n    for split in VALID_SPLITS\n    for path in (source_image_root / split).glob(\"*.png\")\n}\nactual_label_keys = {\n    (path.parent.name, path.stem)\n    for split in VALID_SPLITS\n    for path in (source_label_root / split).glob(\"*.txt\")\n}\nextra_preprocessed_images = sorted(actual_image_keys - expected_image_keys)\nextra_preprocessed_labels = sorted(actual_label_keys - expected_image_keys)\npd.DataFrame(\n    extra_preprocessed_images, columns=[\"split\", \"image_id\"]\n).to_csv(REPORT_ROOT / \"extra_preprocessed_images.csv\", index=False)\npd.DataFrame(\n    extra_preprocessed_labels, columns=[\"split\", \"image_id\"]\n).to_csv(REPORT_ROOT / \"extra_preprocessed_labels.csv\", index=False)\n\nunexpected_source_files = []\nfor split in VALID_SPLITS:\n    unexpected_source_files.extend(\n        str(path)\n        for path in (source_image_root / split).iterdir()\n        if path.is_file() and path.suffix.lower() != \".png\"\n    )\n    unexpected_source_files.extend(\n        str(path)\n        for path in (source_label_root / split).iterdir()\n        if path.is_file() and path.suffix.lower() != \".txt\"\n    )\npd.DataFrame({\"path\": unexpected_source_files}).to_csv(\n    REPORT_ROOT / \"unexpected_source_files.csv\", index=False\n)\n\ndicom_inside_selected_dataset = [\n    path\n    for path in source_dataset_root.rglob(\"*\")\n    if path.is_file() and path.suffix.lower() in {\".dcm\", \".dicom\"}\n]\nif dicom_inside_selected_dataset:\n    raise RuntimeError(\n        \"选中的数据集内部发现 DICOM，说明它不是纯预处理输出：\"\n        f\"{dicom_inside_selected_dataset[0]}\"\n    )\n\nsource_yaml = yaml.safe_load(\n    source_data_yaml_path.read_text(encoding=\"utf-8\")\n)\nsource_yaml_names = source_yaml.get(\"names\", {})\nif isinstance(source_yaml_names, list):\n    source_yaml_class_name = (\n        source_yaml_names[YOLO_CLASS_ID]\n        if len(source_yaml_names) > YOLO_CLASS_ID else None\n    )\nelse:\n    source_yaml_class_name = source_yaml_names.get(\n        YOLO_CLASS_ID, source_yaml_names.get(str(YOLO_CLASS_ID))\n    )\nsource_yaml_class_matches = source_yaml_class_name == TARGET_CLASS\nif not source_yaml_class_matches:\n    raise RuntimeError(\n        \"源 data.yaml 类别与 Nodule/Mass 锁定目标不一致。\"\n    )\n\nsource_baseline_hash_column = next(\n    (\n        column\n        for column in (\"image_sha256\", \"sha256\", \"png_sha256\")\n        if column in manifest.columns\n    ),\n    None,\n)\n\nprint(\"Manifest images:\", len(manifest))\nprint(\"Source count contract:\", source_count_contract_pass)\nprint(\"Source data.yaml class:\", source_yaml_class_name)\nprint(manifest.groupby(\"split\").size())\nprint(\"Geometry manifest contract:\", geometry_manifest_contract_pass)\nprint(\"Missing PNG/label pairs:\", len(missing_source_files))\nprint(\"Source baseline hash column:\", source_baseline_hash_column)\nprint(\"DICOM files read:\", 0)\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. 清洗融合框并验证几何\n\n融合框仍处于原始 DICOM 坐标系。V6 先验证 640version1 的等比例缩放\n契约，再用原图尺寸生成归一化 YOLO 坐标。后续还会把真正写入 TXT 的\n8 位小数坐标投影回已解码 PNG，核对每个框的输出像素误差。\n","metadata":{}},{"cell_type":"code","source":"source_fused_boxes = pd.read_csv(source_fused_boxes_path)\nmissing_fused_columns = REQUIRED_FUSED_COLUMNS - set(source_fused_boxes.columns)\nif missing_fused_columns:\n    raise ValueError(\n        f\"预处理融合框报告缺少字段：{sorted(missing_fused_columns)}\"\n    )\n\nsource_fused_boxes[\"image_id\"] = (\n    source_fused_boxes[\"image_id\"].astype(str).str.strip()\n)\ninvalid_fused_image_ids = source_fused_boxes.loc[\n    source_fused_boxes[\"image_id\"].str.lower().isin(\n        [\"\", \"nan\", \"none\", \"null\"]\n    )\n].copy()\ninvalid_fused_image_ids.to_csv(\n    REPORT_ROOT / \"invalid_fused_image_ids.csv\", index=False\n)\nif len(invalid_fused_image_ids):\n    raise RuntimeError(\"融合框报告含空或非法 image_id。\")\nif len(source_fused_boxes) != EXPECTED_SOURCE_FUSED_BOX_COUNT:\n    raise RuntimeError(\n        \"源融合框数量不是锁定版本的 \"\n        f\"{EXPECTED_SOURCE_FUSED_BOX_COUNT}，实际为 \"\n        f\"{len(source_fused_boxes)}。\"\n    )\nsource_fused_boxes[\"_source_row\"] = source_fused_boxes.index.astype(int)\nsource_fused_boxes[COORDINATE_COLUMNS] = (\n    source_fused_boxes[COORDINATE_COLUMNS].apply(\n        pd.to_numeric, errors=\"coerce\"\n    )\n)\nsource_positive_ids = set(source_fused_boxes[\"image_id\"])\nmanifest_ids = set(manifest[\"image_id\"])\nsource_fused_count_by_image = (\n    source_fused_boxes.groupby(\"image_id\").size()\n    .rename(\"source_fused_box_count\")\n)\nsource_manifest_box_contract = manifest[\n    [\"image_id\", \"is_positive\", \"label_count\"]\n].merge(\n    source_fused_count_by_image,\n    on=\"image_id\",\n    how=\"left\",\n    validate=\"one_to_one\",\n)\nsource_manifest_box_contract[\"source_fused_box_count\"] = (\n    source_manifest_box_contract[\"source_fused_box_count\"]\n    .fillna(0).astype(int)\n)\nsource_manifest_box_contract[\"count_matches\"] = (\n    source_manifest_box_contract[\"label_count\"].astype(int)\n    == source_manifest_box_contract[\"source_fused_box_count\"]\n)\nsource_manifest_box_contract[\"status_matches\"] = (\n    source_manifest_box_contract[\"is_positive\"].astype(int)\n    == (\n        source_manifest_box_contract[\"source_fused_box_count\"] > 0\n    ).astype(int)\n)\nsource_manifest_box_contract_failures = (\n    source_manifest_box_contract.loc[\n        ~source_manifest_box_contract[\"count_matches\"]\n        | ~source_manifest_box_contract[\"status_matches\"]\n    ].copy()\n)\nsource_manifest_box_contract.to_csv(\n    REPORT_ROOT / \"source_manifest_box_contract.csv\", index=False\n)\nsource_manifest_box_contract_failures.to_csv(\n    REPORT_ROOT / \"source_manifest_box_contract_failures.csv\",\n    index=False,\n)\nsource_manifest_box_contract_pass = (\n    len(source_manifest_box_contract_failures) == 0\n)\n\nbox_table = source_fused_boxes.merge(\n    manifest[\n        [\n            \"image_id\", \"original_width\", \"original_height\",\n            \"output_width\", \"output_height\",\n        ]\n    ],\n    on=\"image_id\",\n    how=\"left\",\n    validate=\"many_to_one\",\n)\nfor coordinate in COORDINATE_COLUMNS:\n    box_table[f\"raw_{coordinate}\"] = box_table[coordinate]\n\nissue_records = []\n\n\ndef record_issues(frame, mask, issue_type, action, detail):\n    if not bool(mask.any()):\n        return\n    preferred = [\n        \"_source_row\", \"image_id\",\n        *[f\"raw_{column}\" for column in COORDINATE_COLUMNS],\n        *COORDINATE_COLUMNS,\n        \"original_width\", \"original_height\", \"output_width\", \"output_height\",\n        \"visible_fraction\", \"output_box_width\", \"output_box_height\",\n        \"radiologist_count\", \"source_box_count\", \"radiologist_ids\",\n    ]\n    columns = [column for column in preferred if column in frame.columns]\n    issue = frame.loc[mask, columns].copy()\n    issue[\"issue_type\"] = issue_type\n    issue[\"action\"] = action\n    issue[\"issue_detail\"] = detail\n    issue_records.append(issue)\n\n\nnumeric_ok = np.isfinite(box_table[COORDINATE_COLUMNS]).all(axis=1)\nrecord_issues(\n    box_table, ~numeric_ok, \"non_finite_coordinate\", \"dropped\",\n    \"At least one fused coordinate is NaN, inf, or non-numeric.\",\n)\nbox_table = box_table.loc[numeric_ok].copy()\n\nmissing_dimensions = box_table[\n    [\"original_width\", \"original_height\", \"output_width\", \"output_height\"]\n].isna().any(axis=1)\nrecord_issues(\n    box_table, missing_dimensions, \"missing_manifest_dimensions\", \"dropped\",\n    \"No matching preprocessed manifest row.\",\n)\nbox_table = box_table.loc[~missing_dimensions].copy()\n\npositive_area = (\n    (box_table[\"x_max\"] > box_table[\"x_min\"])\n    & (box_table[\"y_max\"] > box_table[\"y_min\"])\n)\nrecord_issues(\n    box_table, ~positive_area, \"non_positive_area\", \"dropped\",\n    \"x_max <= x_min or y_max <= y_min.\",\n)\nbox_table = box_table.loc[positive_area].copy()\n\nbox_table[\"raw_box_width\"] = box_table[\"x_max\"] - box_table[\"x_min\"]\nbox_table[\"raw_box_height\"] = box_table[\"y_max\"] - box_table[\"y_min\"]\nbox_table[\"raw_box_area\"] = (\n    box_table[\"raw_box_width\"] * box_table[\"raw_box_height\"]\n)\n\ncompletely_outside = (\n    (box_table[\"x_max\"] <= 0)\n    | (box_table[\"y_max\"] <= 0)\n    | (box_table[\"x_min\"] >= box_table[\"original_width\"])\n    | (box_table[\"y_min\"] >= box_table[\"original_height\"])\n)\nrecord_issues(\n    box_table, completely_outside, \"box_outside_image\", \"dropped\",\n    \"Fused box has no intersection with the original image.\",\n)\nbox_table = box_table.loc[~completely_outside].copy()\n\nbox_table[\"clipped_x_min\"] = box_table[\"x_min\"].clip(lower=0)\nbox_table[\"clipped_y_min\"] = box_table[\"y_min\"].clip(lower=0)\nbox_table[\"clipped_x_max\"] = np.minimum(\n    box_table[\"x_max\"], box_table[\"original_width\"]\n)\nbox_table[\"clipped_y_max\"] = np.minimum(\n    box_table[\"y_max\"], box_table[\"original_height\"]\n)\nbox_table[\"clipped_box_width\"] = (\n    box_table[\"clipped_x_max\"] - box_table[\"clipped_x_min\"]\n)\nbox_table[\"clipped_box_height\"] = (\n    box_table[\"clipped_y_max\"] - box_table[\"clipped_y_min\"]\n)\nbox_table[\"clipped_box_area\"] = (\n    box_table[\"clipped_box_width\"] * box_table[\"clipped_box_height\"]\n)\nbox_table[\"visible_fraction\"] = (\n    box_table[\"clipped_box_area\"] / box_table[\"raw_box_area\"]\n)\n\nlow_visible_fraction = (\n    box_table[\"visible_fraction\"] < MIN_VISIBLE_FRACTION\n)\nrecord_issues(\n    box_table,\n    low_visible_fraction,\n    \"insufficient_visible_fraction\",\n    \"dropped\",\n    f\"Visible box fraction is below {MIN_VISIBLE_FRACTION:.2f}.\",\n)\nbox_table = box_table.loc[~low_visible_fraction].copy()\n\npartially_outside = (\n    (box_table[\"x_min\"] < 0)\n    | (box_table[\"y_min\"] < 0)\n    | (box_table[\"x_max\"] > box_table[\"original_width\"])\n    | (box_table[\"y_max\"] > box_table[\"original_height\"])\n)\nrecord_issues(\n    box_table,\n    partially_outside,\n    \"box_clipped_to_boundary\",\n    \"clipped_for_review\",\n    (\n        \"Box was clipped only after passing the visible-fraction threshold; \"\n        \"review before publication.\"\n    ),\n)\nfor coordinate in COORDINATE_COLUMNS:\n    box_table[coordinate] = box_table[f\"clipped_{coordinate}\"]\n\nbox_table[\"output_box_width\"] = (\n    (box_table[\"x_max\"] - box_table[\"x_min\"])\n    * box_table[\"output_width\"]\n    / box_table[\"original_width\"]\n)\nbox_table[\"output_box_height\"] = (\n    (box_table[\"y_max\"] - box_table[\"y_min\"])\n    * box_table[\"output_height\"]\n    / box_table[\"original_height\"]\n)\ntoo_small = (\n    ((box_table[\"x_max\"] - box_table[\"x_min\"]) < MIN_BOX_SIDE_ORIGINAL_PX)\n    | ((box_table[\"y_max\"] - box_table[\"y_min\"]) < MIN_BOX_SIDE_ORIGINAL_PX)\n    | (box_table[\"output_box_width\"] < MIN_BOX_SIDE_OUTPUT_PX)\n    | (box_table[\"output_box_height\"] < MIN_BOX_SIDE_OUTPUT_PX)\n)\nrecord_issues(\n    box_table, too_small, \"box_too_small_after_preprocessing\", \"dropped\",\n    (\n        f\"Box side is below {MIN_BOX_SIDE_ORIGINAL_PX} original pixels or \"\n        f\"{MIN_BOX_SIDE_OUTPUT_PX} output pixels.\"\n    ),\n)\nbox_table = box_table.loc[~too_small].copy()\n\nduplicate_subset = [\"image_id\", *COORDINATE_COLUMNS]\nexact_duplicate_mask = box_table.duplicated(duplicate_subset, keep=\"first\")\nrecord_issues(\n    box_table, exact_duplicate_mask, \"exact_duplicate_fused_box\", \"dropped\",\n    \"Exact duplicate fused box; first occurrence retained.\",\n)\nbox_table = box_table.loc[~exact_duplicate_mask].copy()\n\nclean_fused_boxes = box_table.reset_index(drop=True)\nclean_positive_ids = set(clean_fused_boxes[\"image_id\"])\nambiguous_ids = sorted(source_positive_ids - clean_positive_ids)\npositive_ids_missing_manifest = sorted(source_positive_ids - manifest_ids)\n\nbox_issues = (\n    pd.concat(issue_records, ignore_index=True)\n    if issue_records\n    else pd.DataFrame(\n        columns=[\n            \"_source_row\", \"image_id\", *COORDINATE_COLUMNS,\n            \"issue_type\", \"action\", \"issue_detail\",\n        ]\n    )\n)\nbox_issues.to_csv(\n    REPORT_ROOT / \"preprocessed_fused_box_issues.csv\", index=False\n)\nclean_fused_boxes.to_csv(\n    REPORT_ROOT / \"cleaned_fused_boxes_preprocessed_only.csv\", index=False\n)\nbox_issues.loc[\n    box_issues[\"action\"].eq(\"clipped_for_review\")\n].to_csv(REPORT_ROOT / \"clipped_box_review.csv\", index=False)\nradiologist_metadata_valid = True\nradiologist_metadata_issues = clean_fused_boxes.iloc[0:0].copy()\nif \"radiologist_count\" in clean_fused_boxes.columns:\n    radiologist_counts = pd.to_numeric(\n        clean_fused_boxes[\"radiologist_count\"], errors=\"coerce\"\n    )\n    invalid_radiologist_mask = (\n        ~np.isfinite(radiologist_counts)\n        | (radiologist_counts < 1)\n        | (radiologist_counts\n           > EXPECTED_TRAIN_RADIOLOGISTS_PER_IMAGE)\n        | (radiologist_counts != np.floor(radiologist_counts))\n    )\n    radiologist_metadata_issues = clean_fused_boxes.loc[\n        invalid_radiologist_mask\n    ].copy()\n    radiologist_metadata_valid = len(radiologist_metadata_issues) == 0\n    low_consensus_boxes = clean_fused_boxes.loc[\n        radiologist_counts.fillna(0) < 2\n    ].copy()\nelse:\n    low_consensus_boxes = clean_fused_boxes.iloc[0:0].copy()\nradiologist_metadata_issues.to_csv(\n    REPORT_ROOT / \"invalid_radiologist_metadata.csv\", index=False\n)\nlow_consensus_boxes.to_csv(\n    REPORT_ROOT / \"single_radiologist_box_review.csv\", index=False\n)\nif not radiologist_metadata_valid:\n    raise RuntimeError(\n        \"融合框的 radiologist_count 超出 VinDr-CXR 训练集 1–3 范围。\"\n    )\npd.DataFrame({\"image_id\": ambiguous_ids}).to_csv(\n    REPORT_ROOT / \"ambiguous_positive_images_excluded.csv\", index=False\n)\npd.DataFrame({\"image_id\": positive_ids_missing_manifest}).to_csv(\n    REPORT_ROOT / \"positive_images_missing_manifest.csv\", index=False\n)\n\n# 代数缩放预检查；真正的 TXT 序列化回投影在后续逐框完成。\ngeometry_box_audit = clean_fused_boxes[\n    [\n        \"image_id\", *COORDINATE_COLUMNS,\n        \"original_width\", \"original_height\", \"output_width\", \"output_height\",\n    ]\n].copy()\ncoordinate_axes = {\n    \"x_min\": (\"original_width\", \"output_width\"),\n    \"x_max\": (\"original_width\", \"output_width\"),\n    \"y_min\": (\"original_height\", \"output_height\"),\n    \"y_max\": (\"original_height\", \"output_height\"),\n}\nerror_columns = []\nfor coordinate, (original_dimension, output_dimension) in coordinate_axes.items():\n    normalized = (\n        geometry_box_audit[coordinate]\n        / geometry_box_audit[original_dimension]\n    )\n    projected = normalized * geometry_box_audit[output_dimension]\n    explicitly_scaled = (\n        geometry_box_audit[coordinate]\n        * geometry_box_audit[output_dimension]\n        / geometry_box_audit[original_dimension]\n    )\n    error_column = f\"{coordinate}_roundtrip_error_px\"\n    geometry_box_audit[error_column] = (\n        projected - explicitly_scaled\n    ).abs()\n    error_columns.append(error_column)\ngeometry_box_audit[\"max_roundtrip_error_px\"] = (\n    geometry_box_audit[error_columns].max(axis=1)\n    if len(geometry_box_audit)\n    else pd.Series(dtype=float)\n)\ngeometry_box_audit.to_csv(\n    REPORT_ROOT / \"geometry_box_roundtrip_audit.csv\", index=False\n)\nmax_geometry_roundtrip_error = (\n    float(geometry_box_audit[\"max_roundtrip_error_px\"].max())\n    if len(geometry_box_audit)\n    else 0.0\n)\ngeometry_roundtrip_pass = (\n    np.isfinite(max_geometry_roundtrip_error)\n    and max_geometry_roundtrip_error <= ALGEBRAIC_GEOMETRY_TOLERANCE_PX\n)\nif not geometry_roundtrip_pass:\n    raise RuntimeError(\n        \"原始坐标归一化与实际输出 PNG 投影不一致，禁止生成训练标签。\"\n    )\n\nprint(\"Source fused boxes:\", len(source_fused_boxes))\nprint(\"Source manifest/box contract:\",\n      source_manifest_box_contract_pass)\nprint(\"Radiologist metadata valid:\", radiologist_metadata_valid)\nprint(\"Clean fused boxes:\", len(clean_fused_boxes))\nprint(\"Box issue records:\", len(box_issues))\nprint(\"Ambiguous positive images excluded:\", len(ambiguous_ids))\nprint(\"Single-radiologist fused boxes for review:\", len(low_consensus_boxes))\nprint(\"Max geometry roundtrip error (px):\", max_geometry_roundtrip_error)\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 6. 建立进入自动去重前的候选 manifest 与基础人工复核队列\nusable_manifest = manifest.loc[\n    ~manifest[\"image_id\"].isin(ambiguous_ids)\n].copy()\nusable_manifest[\"is_positive_clean\"] = (\n    usable_manifest[\"image_id\"].isin(clean_positive_ids).astype(int)\n)\n\nsource_box_counts = (\n    source_fused_boxes.groupby(\"image_id\").size().rename(\"source_box_count\")\n)\nclean_box_counts = (\n    clean_fused_boxes.groupby(\"image_id\").size().rename(\"clean_box_count\")\n)\ndropped_issue_summary = (\n    box_issues.loc[box_issues[\"action\"].eq(\"dropped\")]\n    .groupby(\"image_id\")[\"issue_type\"]\n    .agg(lambda values: \"|\".join(sorted(set(values))))\n    .rename(\"drop_reasons\")\n)\nmanual_review_queue = pd.DataFrame({\"image_id\": ambiguous_ids})\nif len(manual_review_queue):\n    manual_review_queue = (\n        manual_review_queue\n        .merge(\n            manifest[[\"image_id\", \"split\", \"is_positive\"]],\n            on=\"image_id\", how=\"left\",\n        )\n        .merge(source_box_counts, on=\"image_id\", how=\"left\")\n        .merge(clean_box_counts, on=\"image_id\", how=\"left\")\n        .merge(dropped_issue_summary, on=\"image_id\", how=\"left\")\n    )\n    manual_review_queue[\"clean_box_count\"] = (\n        manual_review_queue[\"clean_box_count\"].fillna(0).astype(int)\n    )\n    manual_review_queue[\"review_action\"] = (\n        \"Keep excluded; ask a qualified reviewer to confirm/relabel. \"\n        \"Do not automatically mark negative.\"\n    )\n\nsource_positive_ids_in_manifest = source_positive_ids & manifest_ids\nambiguous_ids_in_manifest = set(ambiguous_ids) & manifest_ids\nambiguous_positive_fraction = (\n    len(ambiguous_ids_in_manifest) / len(source_positive_ids_in_manifest)\n    if source_positive_ids_in_manifest\n    else 0.0\n)\n\nprint(\"Candidates before content deduplication:\", len(usable_manifest))\nprint(\"Manual-review images before duplicate checks:\", len(manual_review_queue))\nprint(\"Ambiguous positive fraction:\", round(ambiguous_positive_fraction, 6))\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 7. 根据清洗后的融合框重新生成 YOLO 标签\nFUSED_BOX_MAP = {\n    image_id: group.reset_index(drop=True)\n    for image_id, group in clean_fused_boxes.groupby(\"image_id\", sort=False)\n}\n\n\ndef yolo_lines_for_image(image_id, original_width, original_height):\n    rows = FUSED_BOX_MAP.get(str(image_id))\n    if rows is None or rows.empty:\n        return []\n    lines_out = []\n    for row in rows.itertuples(index=False):\n        # 框处于原始坐标系；640version1 仅等比例缩放、不裁剪、不补边。\n        x_center = ((row.x_min + row.x_max) / 2.0) / original_width\n        y_center = ((row.y_min + row.y_max) / 2.0) / original_height\n        width = (row.x_max - row.x_min) / original_width\n        height = (row.y_max - row.y_min) / original_height\n        values = np.array([x_center, y_center, width, height], dtype=float)\n        if not np.isfinite(values).all():\n            raise ValueError(f\"{image_id}: regenerated non-finite label\")\n        if (\n            width <= 0\n            or height <= 0\n            or x_center - width / 2 < -1e-7\n            or y_center - height / 2 < -1e-7\n            or x_center + width / 2 > 1 + 1e-7\n            or y_center + height / 2 > 1 + 1e-7\n        ):\n            raise ValueError(f\"{image_id}: regenerated box outside image\")\n        values_text = [\n            f\"{value:.{LABEL_SERIALIZATION_DECIMALS}f}\"\n            for value in (x_center, y_center, width, height)\n        ]\n        lines_out.append(\n            f\"{YOLO_CLASS_ID} \" + \" \".join(values_text)\n        )\n    return sorted(lines_out)\n\n\ndef parse_label(path):\n    boxes, errors = [], []\n    if not path.exists():\n        return boxes, [\"missing_label_file\"]\n    for line_number, line in enumerate(\n        path.read_text(encoding=\"utf-8\").splitlines(), start=1\n    ):\n        if not line.strip():\n            continue\n        parts = line.split()\n        if len(parts) != 5:\n            errors.append(f\"line_{line_number}:expected_5_fields\")\n            continue\n        try:\n            class_id = int(parts[0])\n            x_center, y_center, width, height = map(float, parts[1:])\n        except Exception:\n            errors.append(f\"line_{line_number}:non_numeric\")\n            continue\n        values = np.array([x_center, y_center, width, height], dtype=float)\n        if class_id != YOLO_CLASS_ID:\n            errors.append(f\"line_{line_number}:wrong_class_id\")\n        if not np.isfinite(values).all():\n            errors.append(f\"line_{line_number}:non_finite\")\n            continue\n        if width <= 0 or height <= 0:\n            errors.append(f\"line_{line_number}:non_positive_size\")\n        if (\n            x_center - width / 2 < -1e-7\n            or y_center - height / 2 < -1e-7\n            or x_center + width / 2 > 1 + 1e-7\n            or y_center + height / 2 > 1 + 1e-7\n        ):\n            errors.append(f\"line_{line_number}:complete_box_out_of_bounds\")\n        boxes.append((class_id, x_center, y_center, width, height))\n    return boxes, errors\n\n\ndef canonical_boxes(boxes):\n    return sorted(\n        tuple([int(box[0])] + [round(float(value), 6) for value in box[1:]])\n        for box in boxes\n    )\n\n\ndef atomic_write_text(path, text):\n    path.parent.mkdir(parents=True, exist_ok=True)\n    temporary = path.with_suffix(path.suffix + \".tmp\")\n    temporary.write_text(text, encoding=\"utf-8\")\n    temporary.replace(path)\n\n\ndef sha256_file(path, chunk_size=1024 * 1024):\n    digest = hashlib.sha256()\n    with Path(path).open(\"rb\") as handle:\n        for chunk in iter(lambda: handle.read(chunk_size), b\"\"):\n            digest.update(chunk)\n    return digest.hexdigest()\n\n\ndef export_image(source, destination, expected_source_sha256):\n    source = Path(source)\n    destination = Path(destination)\n    current_source_sha256 = sha256_file(source)\n    if current_source_sha256 != expected_source_sha256:\n        raise RuntimeError(\n            \"Source image changed after the pre-export audit.\"\n        )\n    if IMAGE_EXPORT_MODE == \"none\":\n        return None, \"not_exported\"\n    destination.parent.mkdir(parents=True, exist_ok=True)\n    if IMAGE_EXPORT_MODE == \"symlink\":\n        if destination.exists() or destination.is_symlink():\n            destination.unlink()\n        destination.symlink_to(source)\n        return destination, \"symlinked\"\n    if (\n        destination.is_file()\n        and not destination.is_symlink()\n        and destination.stat().st_size == source.stat().st_size\n    ):\n        if (\n            not VERIFY_EXISTING_COPY_HASH\n            or sha256_file(destination) == expected_source_sha256\n        ):\n            return destination, \"verified_existing_copy\"\n    temporary = destination.with_suffix(destination.suffix + \".copying\")\n    shutil.copy2(source, temporary)\n    temporary.replace(destination)\n    return destination, \"copied\"\n\n\n# 7B. 导出前全量解码、文件/像素哈希、坏图审计与精确去重\n# 可变尺寸本身有效；实际尺寸与 manifest 不一致属于几何完整性失败，停止而不删除。\nusable_manifest_before_source_image_cleaning = usable_manifest.copy()\nRESAMPLE_LANCZOS = getattr(Image, \"Resampling\", Image).LANCZOS\n\n\ndef decoded_pixel_sha256(pixels):\n    contiguous = np.ascontiguousarray(pixels, dtype=np.uint8)\n    digest = hashlib.sha256()\n    digest.update(b\"L\")\n    digest.update(int(contiguous.shape[1]).to_bytes(8, \"little\"))\n    digest.update(int(contiguous.shape[0]).to_bytes(8, \"little\"))\n    digest.update(contiguous.tobytes())\n    return digest.hexdigest()\n\n\ndef perceptual_dhash64(grayscale_image):\n    tiny = grayscale_image.resize((9, 8), RESAMPLE_LANCZOS)\n    array = np.asarray(tiny, dtype=np.uint8)\n    bits = array[:, 1:] > array[:, :-1]\n    value = 0\n    for bit in bits.reshape(-1):\n        value = (value << 1) | int(bit)\n    tiny.close()\n    return f\"{value:016x}\"\n\n\ndef normalized_optional_hash(value):\n    if value is None or pd.isna(value) or str(value).strip() == \"\":\n        return None\n    return str(value).strip().lower()\n\n\ndef preexport_source_image_record(row):\n    baseline_value = (\n        getattr(row, source_baseline_hash_column)\n        if source_baseline_hash_column is not None\n        else None\n    )\n    baseline_hash = normalized_optional_hash(baseline_value)\n    record = {\n        \"image_id\": str(row.image_id),\n        \"split\": str(row.split),\n        \"source_image_path\": str(row.source_image_path),\n        \"expected_width\": int(row.output_width),\n        \"expected_height\": int(row.output_height),\n        \"readable\": False,\n        \"format\": None,\n        \"format_is_png\": False,\n        \"mode\": None,\n        \"mode_is_acceptable\": False,\n        \"width\": None,\n        \"height\": None,\n        \"size_matches_manifest\": False,\n        \"geometry_integrity_failure\": False,\n        \"is_constant\": None,\n        \"quality_review_flag\": None,\n        \"source_file_sha256\": None,\n        \"source_pixel_sha256\": None,\n        \"perceptual_dhash64\": None,\n        \"source_baseline_available\": baseline_hash is not None,\n        \"source_baseline_sha256\": baseline_hash,\n        \"source_baseline_hash_matches\": None,\n        \"auto_excludable\": False,\n        \"auto_exclusion_reason\": \"\",\n        \"provenance_failure\": False,\n        \"error\": \"\",\n    }\n    reasons = []\n    try:\n        path = Path(row.source_image_path)\n        file_hash = sha256_file(path)\n        record[\"source_file_sha256\"] = file_hash\n        if baseline_hash is not None:\n            record[\"source_baseline_hash_matches\"] = (\n                file_hash == baseline_hash\n            )\n            record[\"provenance_failure\"] = (\n                record[\"source_baseline_hash_matches\"] is not True\n            )\n\n        with Image.open(path) as image:\n            image.verify()\n        with Image.open(path) as image:\n            record[\"format\"] = image.format\n            record[\"format_is_png\"] = str(image.format).upper() == \"PNG\"\n            record[\"mode\"] = image.mode\n            record[\"mode_is_acceptable\"] = (\n                image.mode == \"L\"\n                if REQUIRE_GRAYSCALE\n                else image.mode in {\"L\", \"RGB\"}\n            )\n            record[\"width\"] = int(image.width)\n            record[\"height\"] = int(image.height)\n            record[\"size_matches_manifest\"] = (\n                image.width == int(row.output_width)\n                and image.height == int(row.output_height)\n            )\n            grayscale = image.convert(\"L\")\n            pixels = np.asarray(grayscale, dtype=np.uint8).copy()\n            extrema = grayscale.getextrema()\n            record[\"min_intensity\"] = int(extrema[0])\n            record[\"max_intensity\"] = int(extrema[1])\n            record[\"is_constant\"] = extrema[0] == extrema[1]\n            record[\"pixel_mean\"] = float(pixels.mean())\n            record[\"pixel_std\"] = float(pixels.std())\n            record[\"near_black_fraction\"] = float((pixels <= 1).mean())\n            record[\"near_white_fraction\"] = float((pixels >= 254).mean())\n            record[\"quality_review_flag\"] = bool(\n                record[\"pixel_std\"] < LOW_CONTRAST_STD_REVIEW_THRESHOLD\n                or record[\"near_black_fraction\"]\n                > EXTREME_SATURATION_REVIEW_FRACTION\n                or record[\"near_white_fraction\"]\n                > EXTREME_SATURATION_REVIEW_FRACTION\n            )\n            record[\"source_pixel_sha256\"] = decoded_pixel_sha256(pixels)\n            record[\"perceptual_dhash64\"] = perceptual_dhash64(grayscale)\n            grayscale.close()\n            record[\"readable\"] = True\n\n        if not record[\"format_is_png\"]:\n            reasons.append(\"not_png\")\n        if not record[\"mode_is_acceptable\"]:\n            reasons.append(\"unexpected_image_mode\")\n        if not record[\"size_matches_manifest\"]:\n            record[\"geometry_integrity_failure\"] = True\n        if record[\"is_constant\"]:\n            reasons.append(\"constant_image\")\n    except Exception as error:\n        record[\"error\"] = f\"{type(error).__name__}: {error}\"\n        reasons.append(\"unreadable_or_corrupt\")\n\n    record[\"auto_exclusion_reason\"] = \"|\".join(sorted(set(reasons)))\n    record[\"auto_excludable\"] = bool(reasons)\n    return record\n\n\npreexport_source_rows = list(\n    usable_manifest_before_source_image_cleaning.itertuples(index=False)\n)\nwith ThreadPoolExecutor(max_workers=AUDIT_WORKERS) as executor:\n    source_image_audit_records = list(\n        tqdm(\n            executor.map(\n                preexport_source_image_record,\n                preexport_source_rows,\n            ),\n            total=len(preexport_source_rows),\n            desc=\"Auditing source PNGs before export\",\n        )\n    )\nsource_image_audit = pd.DataFrame(source_image_audit_records)\nsource_image_audit.to_csv(\n    REPORT_ROOT / \"preexport_source_image_audit.csv\", index=False\n)\n\nsource_geometry_integrity_failures = source_image_audit.loc[\n    source_image_audit[\"geometry_integrity_failure\"].eq(True)\n].copy()\nsource_geometry_integrity_failures.to_csv(\n    REPORT_ROOT / \"source_geometry_integrity_failures.csv\", index=False\n)\nif len(source_geometry_integrity_failures):\n    raise RuntimeError(\n        \"源 PNG 实际尺寸与 manifest 不一致；这是图片—标签几何完整性失败。\"\n        \"程序不会删除这些图片，请修复 manifest 或重新预处理。\"\n    )\n\nsource_provenance_failures = source_image_audit.loc[\n    source_image_audit[\"provenance_failure\"].eq(True)\n].copy()\nsource_provenance_failures.to_csv(\n    REPORT_ROOT / \"source_baseline_hash_failures.csv\", index=False\n)\nif len(source_provenance_failures):\n    raise RuntimeError(\n        \"源 PNG 与 640version1 manifest 中已有哈希不一致；\"\n        \"这属于来源完整性失败，不能靠删除图片继续训练。\"\n    )\n\nautomatic_source_image_exclusions = source_image_audit.loc[\n    source_image_audit[\"auto_excludable\"].eq(True)\n].copy()\nautomatic_source_image_exclusions = automatic_source_image_exclusions.merge(\n    manifest[[\"image_id\", \"is_positive\", \"label_count\"]],\n    on=\"image_id\",\n    how=\"left\",\n    validate=\"one_to_one\",\n)\nautomatic_source_image_exclusions.to_csv(\n    REPORT_ROOT / \"automatic_source_image_exclusions.csv\", index=False\n)\nautomatic_source_excluded_ids = set(\n    automatic_source_image_exclusions[\"image_id\"].astype(str)\n)\nautomatic_source_image_exclusion_fraction = (\n    len(automatic_source_image_exclusions)\n    / len(usable_manifest_before_source_image_cleaning)\n    if len(usable_manifest_before_source_image_cleaning)\n    else 0.0\n)\nautomatic_positive_exclusion_count = int(\n    automatic_source_image_exclusions[\"is_positive\"].fillna(0).sum()\n)\nautomatic_positive_exclusion_fraction = (\n    automatic_positive_exclusion_count\n    / int(manifest[\"is_positive\"].sum())\n    if int(manifest[\"is_positive\"].sum())\n    else 0.0\n)\nautomatic_source_image_exclusion_within_limit = (\n    automatic_source_image_exclusion_fraction\n    <= MAX_AUTOMATIC_IMAGE_EXCLUSION_FRACTION\n    and automatic_positive_exclusion_fraction\n    <= MAX_AUTOMATIC_POSITIVE_EXCLUSION_FRACTION\n)\nif (\n    len(automatic_source_image_exclusions)\n    and not AUTO_EXCLUDE_UNREADABLE_OR_INVALID_IMAGES\n):\n    raise RuntimeError(\n        \"发现损坏或违反 PNG 契约的源图，但自动隔离被关闭。\"\n    )\nif not automatic_source_image_exclusion_within_limit:\n    raise RuntimeError(\n        \"需要自动隔离的坏图比例超过安全阈值；\"\n        \"这可能是预处理整体失败，禁止静默丢弃后训练。\"\n    )\n\nusable_manifest = usable_manifest_before_source_image_cleaning.loc[\n    ~usable_manifest_before_source_image_cleaning[\"image_id\"].isin(\n        automatic_source_excluded_ids\n    )\n].copy()\n\nsource_audit_merge = source_image_audit.loc[\n    ~source_image_audit[\"image_id\"].isin(automatic_source_excluded_ids),\n    [\n        \"image_id\", \"source_file_sha256\", \"source_pixel_sha256\",\n        \"perceptual_dhash64\", \"pixel_mean\", \"pixel_std\",\n        \"near_black_fraction\", \"near_white_fraction\",\n        \"quality_review_flag\", \"source_baseline_available\",\n        \"source_baseline_sha256\", \"source_baseline_hash_matches\",\n    ],\n].rename(columns={\"source_file_sha256\": \"source_sha256_preexport\"})\nusable_manifest = usable_manifest.merge(\n    source_audit_merge,\n    on=\"image_id\",\n    how=\"left\",\n    validate=\"one_to_one\",\n)\n\npreexport_source_audit_complete = (\n    len(source_image_audit)\n    == len(usable_manifest_before_source_image_cleaning)\n    and source_image_audit[\"image_id\"].nunique()\n    == len(usable_manifest_before_source_image_cleaning)\n    and set(source_image_audit[\"image_id\"])\n    == set(usable_manifest_before_source_image_cleaning[\"image_id\"])\n)\nkept_source_hashes_complete = bool(\n    usable_manifest[\n        [\"source_sha256_preexport\", \"source_pixel_sha256\"]\n    ].notna().all().all()\n)\n\n# 锁定整个源版本：控制文件 + 逐图文件 SHA-256。\nsource_control_hashes = {\n    \"manifest.csv\": sha256_file(source_manifest_path),\n    \"dataset_config.json\": sha256_file(source_config_path),\n    \"data.yaml\": sha256_file(source_data_yaml_path),\n    \"fused_boxes_original_coordinates.csv\": sha256_file(\n        source_fused_boxes_path\n    ),\n    \"split_plan.csv\": sha256_file(source_split_plan_path),\n}\nsource_label_hash_records = []\nfor row in manifest.sort_values(\n    [\"split\", \"image_id\"]\n).itertuples(index=False):\n    source_label_hash_records.append({\n        \"image_id\": row.image_id,\n        \"split\": row.split,\n        \"source_label_path\": row.source_label_path,\n        \"source_label_sha256\": sha256_file(\n            Path(row.source_label_path)\n        ),\n    })\nsource_label_hash_manifest = pd.DataFrame(source_label_hash_records)\nsource_label_hash_manifest.to_csv(\n    REPORT_ROOT / \"source_label_hash_manifest.csv\", index=False\n)\nsource_fingerprint_digest = hashlib.sha256()\nsource_fingerprint_digest.update(\n    json.dumps(\n        source_control_hashes, sort_keys=True\n    ).encode(\"utf-8\")\n)\nfor row in source_image_audit.sort_values(\n    [\"split\", \"image_id\"]\n).itertuples(index=False):\n    source_fingerprint_digest.update(\n        (\n            f\"{row.image_id}\\t{row.split}\\t\"\n            f\"{row.source_file_sha256 or ''}\\t{row.error}\\n\"\n        ).encode(\"utf-8\")\n    )\nfor row in source_label_hash_manifest.itertuples(index=False):\n    source_fingerprint_digest.update(\n        (\n            f\"{row.image_id}\\t{row.split}\\t\"\n            f\"{row.source_label_sha256}\\n\"\n        ).encode(\"utf-8\")\n    )\nSOURCE_DATASET_FINGERPRINT = source_fingerprint_digest.hexdigest()\n(REPORT_ROOT / \"source_dataset_identity.json\").write_text(\n    json.dumps({\n        \"source_dataset_signature\": computed_signature,\n        \"source_dataset_fingerprint\": SOURCE_DATASET_FINGERPRINT,\n        \"source_control_hashes\": source_control_hashes,\n        \"source_image_count\": len(source_image_audit),\n        \"source_label_count\": len(source_label_hash_manifest),\n    }, ensure_ascii=False, indent=2),\n    encoding=\"utf-8\",\n)\n\n# 用最终重建标签而不是旧 TXT 决定重复样本是否标签一致。\nlabel_signature_records = []\nfor row in usable_manifest.itertuples(index=False):\n    try:\n        final_lines = yolo_lines_for_image(\n            row.image_id,\n            int(row.original_width),\n            int(row.original_height),\n        )\n        label_signature_records.append({\n            \"image_id\": row.image_id,\n            \"final_label_signature\": \"\\n\".join(final_lines),\n            \"final_label_count\": len(final_lines),\n            \"signature_error\": \"\",\n        })\n    except Exception as error:\n        label_signature_records.append({\n            \"image_id\": row.image_id,\n            \"final_label_signature\": None,\n            \"final_label_count\": None,\n            \"signature_error\": f\"{type(error).__name__}: {error}\",\n        })\nlabel_signature_audit = pd.DataFrame(label_signature_records)\nlabel_signature_errors = label_signature_audit.loc[\n    label_signature_audit[\"signature_error\"].ne(\"\")\n].copy()\nlabel_signature_audit.to_csv(\n    REPORT_ROOT / \"preexport_final_label_signatures.csv\", index=False\n)\nlabel_signature_errors.to_csv(\n    REPORT_ROOT / \"preexport_label_signature_errors.csv\", index=False\n)\nif len(label_signature_errors):\n    raise RuntimeError(\n        f\"有 {len(label_signature_errors)} 张图无法生成最终标签签名。\"\n    )\nusable_manifest = usable_manifest.merge(\n    label_signature_audit[\n        [\"image_id\", \"final_label_signature\", \"final_label_count\"]\n    ],\n    on=\"image_id\",\n    how=\"left\",\n    validate=\"one_to_one\",\n)\n\n# 精确重复使用解码像素哈希，不受 PNG 压缩参数或元数据差异影响。\npixel_group_sizes = (\n    usable_manifest.groupby(\"source_pixel_sha256\")[\"image_id\"]\n    .nunique()\n    .rename(\"content_image_count\")\n)\nduplicate_pixel_hashes = set(\n    pixel_group_sizes.loc[pixel_group_sizes > 1].index\n)\nduplicate_candidates = usable_manifest.loc[\n    usable_manifest[\"source_pixel_sha256\"].isin(duplicate_pixel_hashes)\n].copy()\n\npriority_rank = {\n    split: rank for rank, split in enumerate(DUPLICATE_KEEP_PRIORITY)\n}\nduplicate_decision_records = []\nfor pixel_hash, group in duplicate_candidates.groupby(\n    \"source_pixel_sha256\", sort=True\n):\n    group = group.copy()\n    group[\"_priority_rank\"] = group[\"split\"].map(priority_rank)\n    group = group.sort_values(\n        [\"_priority_rank\", \"image_id\"], kind=\"stable\"\n    )\n    labels_match = (\n        group[\"final_label_signature\"].nunique(dropna=False) == 1\n    )\n    if labels_match:\n        kept = group.iloc[0]\n        kept_image_id = str(kept[\"image_id\"])\n        kept_split = str(kept[\"split\"])\n        for row in group.itertuples(index=False):\n            is_kept = str(row.image_id) == kept_image_id\n            duplicate_decision_records.append({\n                \"decoded_pixel_sha256\": pixel_hash,\n                \"source_file_sha256\": row.source_sha256_preexport,\n                \"image_id\": row.image_id,\n                \"split\": row.split,\n                \"source_image_path\": row.source_image_path,\n                \"final_label_count\": int(row.final_label_count),\n                \"final_label_signature\": row.final_label_signature,\n                \"labels_match_within_group\": True,\n                \"kept_image_id\": kept_image_id,\n                \"kept_split\": kept_split,\n                \"action\": (\n                    \"retained_duplicate_representative\"\n                    if is_kept\n                    else (\n                        \"auto_excluded_exact_pixel_duplicate\"\n                        if AUTO_DEDUPLICATE_EXACT_CONTENT\n                        else \"retained_exact_pixel_duplicate_report_only\"\n                    )\n                ),\n                \"reason\": (\n                    \"Identical decoded pixels and identical final labels; \"\n                    + (\n                        \"automatic deduplication enabled with deterministic \"\n                        \"test>val>train priority, then image_id.\"\n                        if AUTO_DEDUPLICATE_EXACT_CONTENT\n                        else \"report only; all image IDs are retained.\"\n                    )\n                ),\n            })\n    else:\n        for row in group.itertuples(index=False):\n            duplicate_decision_records.append({\n                \"decoded_pixel_sha256\": pixel_hash,\n                \"source_file_sha256\": row.source_sha256_preexport,\n                \"image_id\": row.image_id,\n                \"split\": row.split,\n                \"source_image_path\": row.source_image_path,\n                \"final_label_count\": int(row.final_label_count),\n                \"final_label_signature\": row.final_label_signature,\n                \"labels_match_within_group\": False,\n                \"kept_image_id\": \"\",\n                \"kept_split\": \"\",\n                \"action\": \"quarantined_duplicate_label_conflict\",\n                \"reason\": (\n                    \"Identical decoded pixels but conflicting final labels; \"\n                    \"all copies quarantined and training is blocked.\"\n                ),\n            })\n\nduplicate_decision_columns = [\n    \"decoded_pixel_sha256\", \"source_file_sha256\", \"image_id\", \"split\",\n    \"source_image_path\", \"final_label_count\", \"final_label_signature\",\n    \"labels_match_within_group\", \"kept_image_id\", \"kept_split\",\n    \"action\", \"reason\",\n]\nduplicate_decisions = pd.DataFrame(\n    duplicate_decision_records,\n    columns=duplicate_decision_columns,\n)\nauto_duplicate_exclusions = duplicate_decisions.loc[\n    duplicate_decisions[\"action\"].eq(\n        \"auto_excluded_exact_pixel_duplicate\"\n    )\n].copy()\nretained_exact_pixel_duplicates = duplicate_decisions.loc[\n    duplicate_decisions[\"action\"].eq(\n        \"retained_exact_pixel_duplicate_report_only\"\n    )\n].copy()\nduplicate_label_conflicts = duplicate_decisions.loc[\n    duplicate_decisions[\"action\"].eq(\n        \"quarantined_duplicate_label_conflict\"\n    )\n].copy()\nduplicate_decisions.to_csv(\n    REPORT_ROOT / \"preexport_duplicate_content_decisions.csv\", index=False\n)\nauto_duplicate_exclusions.to_csv(\n    REPORT_ROOT / \"auto_duplicate_exclusions.csv\", index=False\n)\nretained_exact_pixel_duplicates.to_csv(\n    REPORT_ROOT / \"retained_exact_pixel_duplicates.csv\", index=False\n)\nduplicate_label_conflicts.to_csv(\n    REPORT_ROOT / \"duplicate_label_conflicts_quarantined.csv\", index=False\n)\n\nauto_duplicate_excluded_ids = (\n    set(auto_duplicate_exclusions[\"image_id\"].astype(str))\n    if AUTO_DEDUPLICATE_EXACT_CONTENT else set()\n)\nduplicate_conflict_ids = set(\n    duplicate_label_conflicts[\"image_id\"].astype(str)\n)\nduplicate_excluded_ids = (\n    auto_duplicate_excluded_ids | duplicate_conflict_ids\n)\nusable_manifest = usable_manifest.loc[\n    ~usable_manifest[\"image_id\"].isin(duplicate_excluded_ids)\n].copy()\n\n# 所有下游对象只使用同一份最终 ID 集。\nfinal_usable_ids = set(usable_manifest[\"image_id\"])\nclean_fused_boxes = clean_fused_boxes.loc[\n    clean_fused_boxes[\"image_id\"].isin(final_usable_ids)\n].copy()\nclean_fused_boxes = clean_fused_boxes.sort_values(\n    [\"image_id\", \"x_min\", \"y_min\", \"x_max\", \"y_max\"],\n    kind=\"stable\",\n).reset_index(drop=True)\nclean_positive_ids = set(clean_fused_boxes[\"image_id\"])\nFUSED_BOX_MAP = {\n    image_id: group.reset_index(drop=True)\n    for image_id, group in clean_fused_boxes.groupby(\n        \"image_id\", sort=False\n    )\n}\nlow_consensus_boxes = low_consensus_boxes.loc[\n    low_consensus_boxes[\"image_id\"].isin(final_usable_ids)\n].copy()\nclean_fused_boxes.to_csv(\n    REPORT_ROOT / \"cleaned_fused_boxes_preprocessed_only.csv\", index=False\n)\nlow_consensus_boxes.to_csv(\n    REPORT_ROOT / \"single_radiologist_box_review.csv\", index=False\n)\n\n# 真正的标签序列化几何检查：8 位小数 TXT -> 实际输出 PNG 像素。\nserialized_geometry_records = []\ngeometry_audit_source = clean_fused_boxes.rename(\n    columns={\"_source_row\": \"source_row_index\"}\n)\nfor row in geometry_audit_source.itertuples(index=False):\n    original_width = float(row.original_width)\n    original_height = float(row.original_height)\n    output_width = float(row.output_width)\n    output_height = float(row.output_height)\n    normalized_values = [\n        ((row.x_min + row.x_max) / 2.0) / original_width,\n        ((row.y_min + row.y_max) / 2.0) / original_height,\n        (row.x_max - row.x_min) / original_width,\n        (row.y_max - row.y_min) / original_height,\n    ]\n    serialized_values = [\n        float(f\"{value:.{LABEL_SERIALIZATION_DECIMALS}f}\")\n        for value in normalized_values\n    ]\n    x_center, y_center, width, height = serialized_values\n    reconstructed = {\n        \"x_min\": (x_center - width / 2.0) * output_width,\n        \"y_min\": (y_center - height / 2.0) * output_height,\n        \"x_max\": (x_center + width / 2.0) * output_width,\n        \"y_max\": (y_center + height / 2.0) * output_height,\n    }\n    intended = {\n        \"x_min\": row.x_min * output_width / original_width,\n        \"y_min\": row.y_min * output_height / original_height,\n        \"x_max\": row.x_max * output_width / original_width,\n        \"y_max\": row.y_max * output_height / original_height,\n    }\n    errors = {\n        coordinate: abs(reconstructed[coordinate] - intended[coordinate])\n        for coordinate in COORDINATE_COLUMNS\n    }\n    serialized_geometry_records.append({\n        \"image_id\": row.image_id,\n        \"source_row\": int(row.source_row_index),\n        **{\n            f\"intended_{key}_output_px\": value\n            for key, value in intended.items()\n        },\n        **{\n            f\"reconstructed_{key}_output_px\": value\n            for key, value in reconstructed.items()\n        },\n        **{\n            f\"{key}_serialization_error_px\": value\n            for key, value in errors.items()\n        },\n        \"max_serialization_error_px\": max(errors.values()),\n    })\nserialized_geometry_audit = pd.DataFrame(serialized_geometry_records)\nserialized_geometry_audit.to_csv(\n    REPORT_ROOT / \"serialized_yolo_geometry_audit.csv\", index=False\n)\nmax_serialized_geometry_error = (\n    float(\n        serialized_geometry_audit[\"max_serialization_error_px\"].max()\n    )\n    if len(serialized_geometry_audit)\n    else 0.0\n)\nserialized_geometry_pass = (\n    np.isfinite(max_serialized_geometry_error)\n    and max_serialized_geometry_error\n    <= SERIALIZED_GEOMETRY_TOLERANCE_OUTPUT_PX\n)\nif not serialized_geometry_pass:\n    raise RuntimeError(\n        \"最终 YOLO 文本序列化后投影到 PNG 的误差超过阈值。\"\n    )\n\nusable_manifest[\"is_positive_clean\"] = (\n    usable_manifest[\"image_id\"].isin(clean_positive_ids).astype(int)\n)\nsource_positive_mismatch = usable_manifest.loc[\n    usable_manifest[\"is_positive\"].astype(int)\n    != usable_manifest[\"is_positive_clean\"]\n].copy()\nsource_positive_mismatch.to_csv(\n    REPORT_ROOT / \"source_vs_clean_positive_mismatch.csv\", index=False\n)\n\nsplit_sets = {\n    split: set(\n        usable_manifest.loc[\n            usable_manifest[\"split\"].eq(split), \"image_id\"\n        ]\n    )\n    for split in VALID_SPLITS\n}\nimage_split_leakage = (\n    (split_sets[\"train\"] & split_sets[\"val\"])\n    | (split_sets[\"train\"] & split_sets[\"test\"])\n    | (split_sets[\"val\"] & split_sets[\"test\"])\n)\n\nif len(duplicate_label_conflicts):\n    conflict_review_queue = duplicate_label_conflicts[\n        [\n            \"image_id\", \"split\", \"final_label_count\",\n            \"decoded_pixel_sha256\",\n        ]\n    ].copy()\n    conflict_review_queue = conflict_review_queue.merge(\n        manifest[[\"image_id\", \"is_positive\"]],\n        on=\"image_id\",\n        how=\"left\",\n        validate=\"one_to_one\",\n    )\n    conflict_review_queue = conflict_review_queue.merge(\n        source_box_counts,\n        on=\"image_id\",\n        how=\"left\",\n    ).merge(\n        clean_box_counts,\n        on=\"image_id\",\n        how=\"left\",\n    )\n    conflict_review_queue[\"drop_reasons\"] = (\n        \"duplicate_pixels_with_conflicting_final_labels\"\n    )\n    conflict_review_queue[\"review_action\"] = (\n        \"Every copy remains quarantined; a qualified reviewer must \"\n        \"resolve the conflict before the training gate can pass.\"\n    )\n    manual_review_queue = pd.concat(\n        [manual_review_queue, conflict_review_queue],\n        ignore_index=True,\n        sort=False,\n    ).drop_duplicates(\"image_id\", keep=\"first\")\nmanual_review_queue.to_csv(\n    REPORT_ROOT / \"manual_review_queue.csv\", index=False\n)\n\nduplicate_conflict_image_fraction = (\n    len(duplicate_conflict_ids)\n    / len(usable_manifest_before_source_image_cleaning)\n    if len(usable_manifest_before_source_image_cleaning)\n    else 0.0\n)\nno_duplicate_label_conflicts = len(duplicate_label_conflicts) == 0\n\nclass_distribution_valid = (\n    bool(clean_positive_ids & final_usable_ids)\n    and bool(final_usable_ids - clean_positive_ids)\n)\nsplit_class_distribution = (\n    usable_manifest.groupby(\"split\")[\"is_positive_clean\"]\n    .agg(image_count=\"count\", positive_images=\"sum\")\n    .reindex(VALID_SPLITS)\n    .reset_index()\n)\nsplit_class_distribution[\"negative_images\"] = (\n    split_class_distribution[\"image_count\"]\n    - split_class_distribution[\"positive_images\"]\n)\nsplit_class_distribution[\"has_both_classes\"] = (\n    (split_class_distribution[\"positive_images\"] > 0)\n    & (split_class_distribution[\"negative_images\"] > 0)\n)\nsplit_class_distribution.to_csv(\n    REPORT_ROOT / \"split_class_distribution.csv\", index=False\n)\nall_splits_have_both_classes = bool(\n    split_class_distribution[\"has_both_classes\"].all()\n)\n\n# 感知哈希只用于疑似近重复复核；绝不据此自动删医学图像。\nperceptual_hash_collision_rows = pd.DataFrame()\ncross_split_perceptual_hash_review = pd.DataFrame()\nif REPORT_PERCEPTUAL_HASH_COLLISIONS:\n    perceptual_stats = (\n        usable_manifest.dropna(subset=[\"perceptual_dhash64\"])\n        .groupby(\"perceptual_dhash64\")\n        .agg(\n            image_count=(\"image_id\", \"nunique\"),\n            split_count=(\"split\", \"nunique\"),\n            exact_pixel_count=(\"source_pixel_sha256\", \"nunique\"),\n        )\n        .reset_index()\n    )\n    perceptual_collision_hashes = set(\n        perceptual_stats.loc[\n            (perceptual_stats[\"image_count\"] > 1)\n            & (perceptual_stats[\"exact_pixel_count\"] > 1),\n            \"perceptual_dhash64\",\n        ]\n    )\n    cross_split_perceptual_hashes = set(\n        perceptual_stats.loc[\n            (perceptual_stats[\"image_count\"] > 1)\n            & (perceptual_stats[\"split_count\"] > 1)\n            & (perceptual_stats[\"exact_pixel_count\"] > 1),\n            \"perceptual_dhash64\",\n        ]\n    )\n    perceptual_hash_collision_rows = usable_manifest.loc[\n        usable_manifest[\"perceptual_dhash64\"].isin(\n            perceptual_collision_hashes\n        )\n    ].sort_values([\"perceptual_dhash64\", \"split\", \"image_id\"])\n    cross_split_perceptual_hash_review = usable_manifest.loc[\n        usable_manifest[\"perceptual_dhash64\"].isin(\n            cross_split_perceptual_hashes\n        )\n    ].sort_values([\"perceptual_dhash64\", \"split\", \"image_id\"])\nperceptual_hash_collision_rows.to_csv(\n    REPORT_ROOT / \"perceptual_hash_collision_review.csv\", index=False\n)\ncross_split_perceptual_hash_review.to_csv(\n    REPORT_ROOT / \"cross_split_perceptual_hash_review.csv\", index=False\n)\n\n# 固定最终顺序并生成与模型训练绑定的数据指纹。\nsplit_sort_rank = {\n    split: rank for rank, split in enumerate(VALID_SPLITS)\n}\nusable_manifest[\"_split_sort_rank\"] = usable_manifest[\"split\"].map(\n    split_sort_rank\n)\nusable_manifest = usable_manifest.sort_values(\n    [\"_split_sort_rank\", \"image_id\"], kind=\"stable\"\n).drop(columns=\"_split_sort_rank\").reset_index(drop=True)\n\nclean_fingerprint_digest = hashlib.sha256()\nclean_fingerprint_digest.update(\n    (\n        SOURCE_DATASET_FINGERPRINT + \"\\n\"\n        + CLEANING_POLICY_SIGNATURE + \"\\n\"\n    ).encode(\"utf-8\")\n)\nfor row in usable_manifest.itertuples(index=False):\n    clean_fingerprint_digest.update(\n        (\n            f\"{row.image_id}\\t{row.split}\\t\"\n            f\"{row.source_pixel_sha256}\\t\"\n            f\"{row.final_label_signature}\\n\"\n        ).encode(\"utf-8\")\n    )\nCLEAN_DATASET_FINGERPRINT = clean_fingerprint_digest.hexdigest()\n\ndeduplication_summary = {\n    \"automatic_deduplication_enabled\": AUTO_DEDUPLICATE_EXACT_CONTENT,\n    \"exact_content_definition\": \"decoded_grayscale_pixels_sha256\",\n    \"keep_priority\": list(DUPLICATE_KEEP_PRIORITY),\n    \"candidate_images_after_source_image_cleaning\": len(\n        usable_manifest_before_source_image_cleaning\n    ) - len(automatic_source_image_exclusions),\n    \"duplicate_content_groups\": int(\n        duplicate_decisions[\"decoded_pixel_sha256\"].nunique()\n        if len(duplicate_decisions) else 0\n    ),\n    \"safe_redundant_copies_excluded\": len(auto_duplicate_exclusions),\n    \"redundant_copies_retained_report_only\": len(\n        retained_exact_pixel_duplicates\n    ),\n    \"label_conflict_images_quarantined\": len(duplicate_label_conflicts),\n    \"label_conflicts_block_training\": BLOCK_ON_DUPLICATE_LABEL_CONFLICT,\n    \"label_conflict_image_fraction\": duplicate_conflict_image_fraction,\n    \"final_usable_images\": len(usable_manifest),\n    \"source_files_modified\": False,\n}\n(REPORT_ROOT / \"deduplication_summary.json\").write_text(\n    json.dumps(deduplication_summary, ensure_ascii=False, indent=2),\n    encoding=\"utf-8\",\n)\n(REPORT_ROOT / \"source_image_exclusion_summary.json\").write_text(\n    json.dumps({\n        \"automatically_excluded_images\": len(\n            automatic_source_image_exclusions\n        ),\n        \"automatically_excluded_positive_images\": (\n            automatic_positive_exclusion_count\n        ),\n        \"image_exclusion_fraction\": (\n            automatic_source_image_exclusion_fraction\n        ),\n        \"positive_exclusion_fraction\": (\n            automatic_positive_exclusion_fraction\n        ),\n        \"within_configured_limits\": (\n            automatic_source_image_exclusion_within_limit\n        ),\n    }, ensure_ascii=False, indent=2),\n    encoding=\"utf-8\",\n)\n\nprint(\"Pre-export source images audited:\", len(source_image_audit))\nprint(\"Source image audit complete:\", preexport_source_audit_complete)\nprint(\"Automatically excluded invalid images:\", len(\n    automatic_source_image_exclusions\n))\nprint(\"Duplicate decoded-pixel groups:\", deduplication_summary[\n    \"duplicate_content_groups\"\n])\nprint(\"Safe redundant copies auto-excluded:\", len(\n    auto_duplicate_exclusions\n))\nprint(\"Exact duplicate copies retained (report only):\", len(\n    retained_exact_pixel_duplicates\n))\nprint(\"Conflicting duplicate images quarantined:\", len(\n    duplicate_label_conflicts\n))\nprint(\"Final usable preprocessed images:\", len(usable_manifest))\nprint(\n    usable_manifest.groupby(\"split\")[\"is_positive_clean\"].agg(\n        [\"count\", \"sum\"]\n    )\n)\nprint(\"Source vs clean positive mismatch:\", len(source_positive_mismatch))\nprint(\"Image-ID split leakage:\", len(image_split_leakage))\nprint(\"Manual-review images:\", len(manual_review_queue))\nprint(\"Cross-split perceptual-hash review rows:\", len(\n    cross_split_perceptual_hash_review\n))\nprint(\"Max serialized geometry error (px):\", max_serialized_geometry_error)\nprint(\"All splits contain positive and negative images:\",\n      all_splits_have_both_classes)\nprint(\"Source dataset fingerprint:\", SOURCE_DATASET_FINGERPRINT)\nprint(\"Clean dataset fingerprint:\", CLEAN_DATASET_FINGERPRINT)\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 8. 清理 V6 旧残留、磁盘预检并原子导出最终数据集\nfor split in VALID_SPLITS:\n    (NEW_LABEL_ROOT / split).mkdir(parents=True, exist_ok=True)\n    if IMAGE_EXPORT_MODE != \"none\":\n        (NEW_IMAGE_ROOT / split).mkdir(parents=True, exist_ok=True)\n\nexpected_preexport_keys = set(\n    zip(usable_manifest[\"split\"], usable_manifest[\"image_id\"])\n)\nstale_output_records = []\nunexpected_output_directories = []\nfor split in VALID_SPLITS:\n    generated_roots = [\n        (NEW_LABEL_ROOT / split, \".txt\", \"label\")\n    ]\n    if IMAGE_EXPORT_MODE != \"none\":\n        generated_roots.append(\n            (NEW_IMAGE_ROOT / split, \".png\", \"image\")\n        )\n    for generated_root, expected_suffix, kind in generated_roots:\n        for generated_path in generated_root.iterdir():\n            if generated_path.is_dir() and not generated_path.is_symlink():\n                unexpected_output_directories.append({\n                    \"split\": split,\n                    \"kind\": kind,\n                    \"path\": str(generated_path),\n                    \"reason\": \"unexpected_subdirectory_not_removed\",\n                })\n                continue\n            key = (split, generated_path.stem)\n            is_expected = (\n                generated_path.suffix.lower() == expected_suffix\n                and key in expected_preexport_keys\n                and not (\n                    IMAGE_EXPORT_MODE == \"copy\"\n                    and generated_path.is_symlink()\n                )\n            )\n            if not is_expected:\n                generated_path.unlink()\n                stale_output_records.append({\n                    \"split\": split,\n                    \"image_id\": generated_path.stem,\n                    \"kind\": kind,\n                    \"path\": str(generated_path),\n                    \"action\": \"removed_stale_v6_output\",\n                })\nstale_output_cleanup = pd.DataFrame(\n    stale_output_records,\n    columns=[\"split\", \"image_id\", \"kind\", \"path\", \"action\"],\n)\nunexpected_output_directories_df = pd.DataFrame(\n    unexpected_output_directories,\n    columns=[\"split\", \"kind\", \"path\", \"reason\"],\n)\nstale_output_cleanup.to_csv(\n    REPORT_ROOT / \"stale_output_cleanup.csv\", index=False\n)\nunexpected_output_directories_df.to_csv(\n    REPORT_ROOT / \"unexpected_output_directories.csv\", index=False\n)\n\n\ndef existing_copy_matches(row):\n    if IMAGE_EXPORT_MODE != \"copy\":\n        return False\n    destination = (\n        NEW_IMAGE_ROOT / row.split / f\"{row.image_id}.png\"\n    )\n    source = Path(row.source_image_path)\n    try:\n        return (\n            destination.is_file()\n            and not destination.is_symlink()\n            and destination.stat().st_size == source.stat().st_size\n            and (\n                not VERIFY_EXISTING_COPY_HASH\n                or sha256_file(destination)\n                == row.source_sha256_preexport\n            )\n        )\n    except Exception:\n        return False\n\n\nsource_image_bytes = sum(\n    Path(path).stat().st_size\n    for path in usable_manifest[\"source_image_path\"]\n)\nestimated_additional_copy_bytes = 0\nif IMAGE_EXPORT_MODE == \"copy\":\n    for row in usable_manifest.itertuples(index=False):\n        if not existing_copy_matches(row):\n            estimated_additional_copy_bytes += Path(\n                row.source_image_path\n            ).stat().st_size\n\ndisk_usage = shutil.disk_usage(OUTPUT_BASE_ROOT)\nrequired_free_bytes = int(\n    estimated_additional_copy_bytes * DISK_SPACE_SAFETY_FACTOR\n    + DISK_SPACE_RESERVE_GB * (1024 ** 3)\n)\ndisk_preflight_pass = (\n    IMAGE_EXPORT_MODE != \"copy\"\n    or disk_usage.free >= required_free_bytes\n)\ndisk_preflight = {\n    \"image_export_mode\": IMAGE_EXPORT_MODE,\n    \"source_image_bytes\": int(source_image_bytes),\n    \"estimated_additional_copy_bytes\": int(\n        estimated_additional_copy_bytes\n    ),\n    \"free_bytes\": int(disk_usage.free),\n    \"required_free_bytes\": int(required_free_bytes),\n    \"passed\": bool(disk_preflight_pass),\n}\n(REPORT_ROOT / \"disk_preflight.json\").write_text(\n    json.dumps(disk_preflight, indent=2), encoding=\"utf-8\"\n)\nif not disk_preflight_pass:\n    raise RuntimeError(\n        \"磁盘空间预检失败。请清理 /kaggle/working；\"\n        \"正式保存仍必须使用 copy。\"\n    )\n\nexport_record_columns = [\n    \"image_id\", \"split\", \"is_positive\", \"label_count\",\n    \"source_image_path\", \"source_label_path\",\n    \"source_baseline_sha256\", \"source_sha256_preexport\",\n    \"source_pixel_sha256\", \"perceptual_dhash64\",\n    \"final_label_signature\", \"image_path\", \"label_path\",\n    \"original_width\", \"original_height\",\n    \"output_width\", \"output_height\", \"image_export_status\",\n]\nexport_failure_columns = [\n    \"image_id\", \"split\", \"error_type\", \"error_message\"\n]\nexport_records = []\nexport_failures = []\nstart_time = perf_counter()\nfor row in tqdm(\n    usable_manifest.itertuples(index=False),\n    total=len(usable_manifest),\n    desc=\"Exporting final clean dataset\",\n):\n    try:\n        source_image = Path(row.source_image_path)\n        destination_image = (\n            NEW_IMAGE_ROOT / row.split / f\"{row.image_id}.png\"\n        )\n        destination_label = (\n            NEW_LABEL_ROOT / row.split / f\"{row.image_id}.txt\"\n        )\n        label_lines = yolo_lines_for_image(\n            row.image_id,\n            int(row.original_width),\n            int(row.original_height),\n        )\n        if int(row.is_positive_clean) == 1 and not label_lines:\n            raise ValueError(\n                \"Clean positive image has no regenerated label.\"\n            )\n        if int(row.is_positive_clean) == 0 and label_lines:\n            raise ValueError(\n                \"Clean negative image unexpectedly has boxes.\"\n            )\n        final_label_text = (\n            \"\\n\".join(label_lines) + (\"\\n\" if label_lines else \"\")\n        )\n        atomic_write_text(destination_label, final_label_text)\n        exported_image, export_status = export_image(\n            source_image,\n            destination_image,\n            row.source_sha256_preexport,\n        )\n        export_records.append({\n            \"image_id\": row.image_id,\n            \"split\": row.split,\n            \"is_positive\": int(bool(label_lines)),\n            \"label_count\": len(label_lines),\n            \"source_image_path\": str(source_image),\n            \"source_label_path\": str(row.source_label_path),\n            \"source_baseline_sha256\": (\n                row.source_baseline_sha256\n            ),\n            \"source_sha256_preexport\": (\n                row.source_sha256_preexport\n            ),\n            \"source_pixel_sha256\": row.source_pixel_sha256,\n            \"perceptual_dhash64\": row.perceptual_dhash64,\n            \"final_label_signature\": row.final_label_signature,\n            \"image_path\": (\n                str(exported_image)\n                if exported_image is not None\n                else str(source_image)\n            ),\n            \"label_path\": str(destination_label),\n            \"original_width\": int(row.original_width),\n            \"original_height\": int(row.original_height),\n            \"output_width\": int(row.output_width),\n            \"output_height\": int(row.output_height),\n            \"image_export_status\": export_status,\n        })\n    except Exception as error:\n        export_failures.append({\n            \"image_id\": row.image_id,\n            \"split\": row.split,\n            \"error_type\": type(error).__name__,\n            \"error_message\": str(error),\n        })\n\nmanifest_clean = pd.DataFrame(\n    export_records, columns=export_record_columns\n)\nexport_failures_df = pd.DataFrame(\n    export_failures, columns=export_failure_columns\n)\nmanifest_clean.to_csv(DATASET_ROOT / \"manifest.csv\", index=False)\nexport_failures_df.to_csv(\n    REPORT_ROOT / \"export_failures.csv\", index=False\n)\n\ndata_yaml_path = DATASET_ROOT / \"data.yaml\"\nif IMAGE_EXPORT_MODE != \"none\":\n    clean_data_config = {\n        # 空 path 由 Ultralytics 解析为 data.yaml 所在目录。\n        \"path\": \"\",\n        \"train\": \"images/train\",\n        \"val\": \"images/val\",\n        \"test\": \"images/test\",\n        \"nc\": 1,\n        \"names\": {YOLO_CLASS_ID: TARGET_CLASS},\n    }\n    atomic_write_text(\n        data_yaml_path,\n        yaml.safe_dump(\n            clean_data_config,\n            sort_keys=False,\n            allow_unicode=True,\n        ),\n    )\n\nexpected_output_keys = set(\n    zip(manifest_clean[\"split\"], manifest_clean[\"image_id\"])\n)\nactual_output_image_keys = (\n    {\n        (path.parent.name, path.stem)\n        for split in VALID_SPLITS\n        for path in (NEW_IMAGE_ROOT / split).glob(\"*.png\")\n    }\n    if IMAGE_EXPORT_MODE != \"none\"\n    else set()\n)\nactual_output_label_keys = {\n    (path.parent.name, path.stem)\n    for split in VALID_SPLITS\n    for path in (NEW_LABEL_ROOT / split).glob(\"*.txt\")\n}\nextra_output_images = sorted(\n    actual_output_image_keys - expected_output_keys\n)\nextra_output_labels = sorted(\n    actual_output_label_keys - expected_output_keys\n)\nmissing_output_images = sorted(\n    expected_output_keys - actual_output_image_keys\n) if IMAGE_EXPORT_MODE != \"none\" else []\nmissing_output_labels = sorted(\n    expected_output_keys - actual_output_label_keys\n)\npd.DataFrame(\n    extra_output_images, columns=[\"split\", \"image_id\"]\n).to_csv(REPORT_ROOT / \"extra_output_images.csv\", index=False)\npd.DataFrame(\n    extra_output_labels, columns=[\"split\", \"image_id\"]\n).to_csv(REPORT_ROOT / \"extra_output_labels.csv\", index=False)\npd.DataFrame(\n    missing_output_images, columns=[\"split\", \"image_id\"]\n).to_csv(REPORT_ROOT / \"missing_output_images.csv\", index=False)\npd.DataFrame(\n    missing_output_labels, columns=[\"split\", \"image_id\"]\n).to_csv(REPORT_ROOT / \"missing_output_labels.csv\", index=False)\n\npostexport_unexpected_entries = []\nfor split in VALID_SPLITS:\n    roots = [(NEW_LABEL_ROOT / split, \".txt\", \"label\")]\n    if IMAGE_EXPORT_MODE != \"none\":\n        roots.append((NEW_IMAGE_ROOT / split, \".png\", \"image\"))\n    for root, suffix, kind in roots:\n        for path in root.iterdir():\n            if (\n                path.is_dir()\n                or (\n                    IMAGE_EXPORT_MODE == \"copy\"\n                    and path.is_symlink()\n                )\n                or path.suffix.lower() != suffix\n                or (split, path.stem) not in expected_output_keys\n            ):\n                postexport_unexpected_entries.append({\n                    \"split\": split,\n                    \"kind\": kind,\n                    \"path\": str(path),\n                })\npostexport_unexpected_entries_df = pd.DataFrame(\n    postexport_unexpected_entries,\n    columns=[\"split\", \"kind\", \"path\"],\n)\npostexport_unexpected_entries_df.to_csv(\n    REPORT_ROOT / \"unexpected_output_entries.csv\", index=False\n)\n\nyaml_paths_exist = False\nyaml_class_matches = False\nif IMAGE_EXPORT_MODE != \"none\" and data_yaml_path.exists():\n    loaded_yaml = yaml.safe_load(\n        data_yaml_path.read_text(encoding=\"utf-8\")\n    )\n    yaml_root = (\n        data_yaml_path.parent\n        if not loaded_yaml.get(\"path\")\n        else Path(loaded_yaml[\"path\"])\n    )\n    yaml_paths_exist = all(\n        (yaml_root / loaded_yaml[split]).is_dir()\n        for split in VALID_SPLITS\n    )\n    yaml_names = loaded_yaml.get(\"names\", {})\n    yaml_class_name = (\n        yaml_names[YOLO_CLASS_ID]\n        if isinstance(yaml_names, list)\n        else yaml_names.get(\n            YOLO_CLASS_ID, yaml_names.get(str(YOLO_CLASS_ID))\n        )\n    )\n    yaml_class_matches = (\n        yaml_class_name == TARGET_CLASS\n        and int(loaded_yaml.get(\"nc\", 1)) == 1\n    )\n\noutput_split_counts = {\n    split: int(count)\n    for split, count in manifest_clean.groupby(\"split\").size().items()\n}\ndataset_identity = {\n    \"cleaning_version\": \"6.0-release\",\n    \"source_dataset_signature\": computed_signature,\n    \"source_dataset_fingerprint\": SOURCE_DATASET_FINGERPRINT,\n    \"cleaning_policy_signature\": CLEANING_POLICY_SIGNATURE,\n    \"clean_dataset_fingerprint\": CLEAN_DATASET_FINGERPRINT,\n    \"target_class\": TARGET_CLASS,\n    \"class_id\": YOLO_CLASS_ID,\n    \"image_count\": int(len(manifest_clean)),\n    \"split_counts\": output_split_counts,\n    \"source_files_modified\": False,\n}\n(DATASET_ROOT / \"dataset_identity.json\").write_text(\n    json.dumps(dataset_identity, ensure_ascii=False, indent=2),\n    encoding=\"utf-8\",\n)\nv6_dataset_config = {\n    **CLEANING_POLICY,\n    **dataset_identity,\n    \"source_dataset_root\": str(source_dataset_root),\n    \"image_export_mode\": IMAGE_EXPORT_MODE,\n    \"audit_mode\": AUDIT_MODE,\n    \"references\": [\n        \"10.1038/s41597-022-01498-w\",\n        \"10.1016/j.patter.2023.100804\",\n    ],\n}\n(DATASET_ROOT / \"cleaning_config_v6.json\").write_text(\n    json.dumps(v6_dataset_config, ensure_ascii=False, indent=2),\n    encoding=\"utf-8\",\n)\n\nprint(\"Disk preflight:\", disk_preflight_pass)\nprint(\"Removed stale V6 output files:\", len(stale_output_cleanup))\nprint(\"Exported images:\", len(manifest_clean))\nprint(\"Export failures:\", len(export_failures_df))\nprint(\"Missing output images/labels:\",\n      len(missing_output_images), len(missing_output_labels))\nprint(\"Extra output images/labels:\",\n      len(extra_output_images), len(extra_output_labels))\nprint(\"Unexpected output entries:\",\n      len(postexport_unexpected_entries_df))\nprint(\"Elapsed seconds:\", round(perf_counter() - start_time, 1))\nif IMAGE_EXPORT_MODE != \"none\":\n    print(\"Portable training YAML:\", data_yaml_path)\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 9. 对比 640version1 源标签和 V6 重建标签\nlabel_audit_records = []\nlabel_comparison_records = []\nfor row in tqdm(\n    manifest_clean.itertuples(index=False),\n    total=len(manifest_clean),\n    desc=\"Comparing source and rebuilt labels\",\n):\n    source_boxes, source_errors = parse_label(Path(row.source_label_path))\n    clean_boxes, clean_errors = parse_label(Path(row.label_path))\n    label_audit_records.extend([\n        {\n            \"version\": \"640version1_source\",\n            \"image_id\": row.image_id,\n            \"split\": row.split,\n            \"path\": row.source_label_path,\n            \"box_count\": len(source_boxes),\n            \"error_count\": len(source_errors),\n            \"errors\": \"|\".join(source_errors),\n        },\n        {\n            \"version\": \"final_v6\",\n            \"image_id\": row.image_id,\n            \"split\": row.split,\n            \"path\": row.label_path,\n            \"box_count\": len(clean_boxes),\n            \"error_count\": len(clean_errors),\n            \"errors\": \"|\".join(clean_errors),\n        },\n    ])\n    label_comparison_records.append({\n        \"image_id\": row.image_id,\n        \"split\": row.split,\n        \"manifest_label_count\": int(row.label_count),\n        \"source_box_count\": len(source_boxes),\n        \"clean_box_count\": len(clean_boxes),\n        \"source_error_count\": len(source_errors),\n        \"clean_error_count\": len(clean_errors),\n        \"same_labels_rounded_6dp\": (\n            canonical_boxes(source_boxes) == canonical_boxes(clean_boxes)\n        ),\n    })\n\nlabel_audit = pd.DataFrame(label_audit_records)\nlabel_comparison = pd.DataFrame(label_comparison_records)\nlabel_audit.to_csv(REPORT_ROOT / \"label_audit.csv\", index=False)\nlabel_comparison.to_csv(\n    REPORT_ROOT / \"source_vs_clean_label_comparison.csv\", index=False\n)\nsource_label_errors = label_audit.loc[\n    label_audit[\"version\"].eq(\"640version1_source\")\n    & (label_audit[\"error_count\"] > 0)\n].copy()\nclean_label_errors = label_audit.loc[\n    label_audit[\"version\"].eq(\"final_v6\")\n    & (label_audit[\"error_count\"] > 0)\n].copy()\nclean_label_count_mismatch = label_comparison.loc[\n    label_comparison[\"manifest_label_count\"]\n    != label_comparison[\"clean_box_count\"]\n].copy()\nsource_label_errors.to_csv(\n    REPORT_ROOT / \"source_label_errors.csv\", index=False\n)\nclean_label_errors.to_csv(\n    REPORT_ROOT / \"clean_label_errors.csv\", index=False\n)\nclean_label_count_mismatch.to_csv(\n    REPORT_ROOT / \"clean_label_count_mismatch.csv\", index=False\n)\n\nprint(\"Invalid source label files:\", len(source_label_errors))\nprint(\"Invalid rebuilt label files:\", len(clean_label_errors))\nprint(\"Clean label-count mismatch:\", len(clean_label_count_mismatch))\nprint(\n    \"Labels changed by cleaning:\",\n    int((~label_comparison[\"same_labels_rounded_6dp\"]).sum()),\n)\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 10. 对最终导出 PNG 再做全量解码、文件哈希和像素哈希审计\ndef audit_exported_image(row):\n    path = Path(row.image_path)\n    baseline_hash = normalized_optional_hash(\n        row.source_baseline_sha256\n    )\n    record = {\n        \"image_id\": str(row.image_id),\n        \"split\": str(row.split),\n        \"path\": str(path),\n        \"source_path\": str(row.source_image_path),\n        \"readable\": False,\n        \"format_is_png\": False,\n        \"size_matches_manifest\": False,\n        \"mode\": None,\n        \"is_constant\": None,\n        \"quality_review_flag\": None,\n        \"source_baseline_available\": baseline_hash is not None,\n        \"source_baseline_hash_matches\": None,\n        \"export_hash_matches_source\": None,\n        \"export_pixel_hash_matches_source\": None,\n        \"error\": \"\",\n    }\n    try:\n        with Image.open(path) as image:\n            image.verify()\n        with Image.open(path) as image:\n            record[\"format\"] = image.format\n            record[\"format_is_png\"] = (\n                str(image.format).upper() == \"PNG\"\n            )\n            record[\"mode\"] = image.mode\n            record[\"width\"] = int(image.width)\n            record[\"height\"] = int(image.height)\n            record[\"size_matches_manifest\"] = (\n                image.width == int(row.output_width)\n                and image.height == int(row.output_height)\n            )\n            grayscale = image.convert(\"L\")\n            pixels = np.asarray(\n                grayscale, dtype=np.uint8\n            ).copy()\n            extrema = grayscale.getextrema()\n            record[\"min_intensity\"] = int(extrema[0])\n            record[\"max_intensity\"] = int(extrema[1])\n            record[\"is_constant\"] = extrema[0] == extrema[1]\n            record[\"pixel_mean\"] = float(pixels.mean())\n            record[\"pixel_std\"] = float(pixels.std())\n            record[\"near_black_fraction\"] = float(\n                (pixels <= 1).mean()\n            )\n            record[\"near_white_fraction\"] = float(\n                (pixels >= 254).mean()\n            )\n            record[\"quality_review_flag\"] = bool(\n                record[\"pixel_std\"]\n                < LOW_CONTRAST_STD_REVIEW_THRESHOLD\n                or record[\"near_black_fraction\"]\n                > EXTREME_SATURATION_REVIEW_FRACTION\n                or record[\"near_white_fraction\"]\n                > EXTREME_SATURATION_REVIEW_FRACTION\n            )\n            export_pixel_hash = decoded_pixel_sha256(pixels)\n            record[\"export_pixel_sha256\"] = export_pixel_hash\n            record[\"export_pixel_hash_matches_source\"] = (\n                export_pixel_hash == row.source_pixel_sha256\n            )\n            record[\"export_perceptual_dhash64\"] = (\n                perceptual_dhash64(grayscale)\n            )\n            grayscale.close()\n            record[\"readable\"] = True\n\n        export_hash = sha256_file(path)\n        record[\"export_sha256\"] = export_hash\n        record[\"source_sha256\"] = row.source_sha256_preexport\n        record[\"source_pixel_sha256\"] = row.source_pixel_sha256\n        record[\"export_hash_matches_source\"] = (\n            export_hash == row.source_sha256_preexport\n        )\n        if baseline_hash is not None:\n            record[\"source_baseline_hash_matches\"] = (\n                export_hash == baseline_hash\n            )\n    except Exception as error:\n        record[\"error\"] = f\"{type(error).__name__}: {error}\"\n    return record\n\n\nif AUDIT_MODE == \"full\":\n    image_audit_source = manifest_clean.copy()\nelse:\n    image_audit_source = pd.concat(\n        [\n            group.sample(\n                n=min(SAMPLE_IMAGES_PER_SPLIT, len(group)),\n                random_state=SEED,\n            )\n            for _, group in manifest_clean.groupby(\"split\")\n        ],\n        ignore_index=True,\n    )\n\naudit_rows = list(image_audit_source.itertuples(index=False))\nwith ThreadPoolExecutor(max_workers=AUDIT_WORKERS) as executor:\n    image_audit_records = list(\n        tqdm(\n            executor.map(audit_exported_image, audit_rows),\n            total=len(audit_rows),\n            desc=f\"Auditing exported PNGs ({AUDIT_MODE})\",\n        )\n    )\nimage_audit = pd.DataFrame(image_audit_records)\n\nif REQUIRE_GRAYSCALE:\n    image_audit[\"mode_is_acceptable\"] = image_audit[\"mode\"].eq(\"L\")\nelse:\n    image_audit[\"mode_is_acceptable\"] = image_audit[\"mode\"].isin(\n        [\"L\", \"RGB\"]\n    )\n\nimage_audit[\"hash_passes\"] = (\n    image_audit[\"export_hash_matches_source\"].eq(True)\n    & image_audit[\"export_pixel_hash_matches_source\"].eq(True)\n    & (\n        ~image_audit[\"source_baseline_available\"]\n        | image_audit[\"source_baseline_hash_matches\"].eq(True)\n    )\n)\nimage_audit[\"passes\"] = (\n    image_audit[\"readable\"]\n    & image_audit[\"format_is_png\"]\n    & image_audit[\"size_matches_manifest\"]\n    & image_audit[\"mode_is_acceptable\"]\n    & ~image_audit[\"is_constant\"].fillna(True)\n    & image_audit[\"hash_passes\"]\n)\nfailed_image_audit = image_audit.loc[\n    ~image_audit[\"passes\"]\n].copy()\nimage_quality_review = image_audit.loc[\n    image_audit[\"quality_review_flag\"].eq(True)\n].copy()\nimage_audit.to_csv(\n    REPORT_ROOT / \"image_audit.csv\", index=False\n)\nfailed_image_audit.to_csv(\n    REPORT_ROOT / \"failed_image_audit.csv\", index=False\n)\nimage_quality_review.to_csv(\n    REPORT_ROOT / \"image_quality_review.csv\", index=False\n)\n\nhash_columns = [\n    \"image_id\", \"split\", \"source_sha256\", \"export_sha256\",\n    \"source_pixel_sha256\", \"export_pixel_sha256\",\n    \"source_baseline_available\", \"source_baseline_hash_matches\",\n    \"export_hash_matches_source\",\n    \"export_pixel_hash_matches_source\",\n]\nimage_hash_manifest = image_audit[\n    [column for column in hash_columns if column in image_audit.columns]\n].copy()\nimage_hash_manifest.to_csv(\n    REPORT_ROOT / \"image_hash_manifest.csv\", index=False\n)\n\nif {\n    \"export_sha256\", \"export_pixel_sha256\"\n}.issubset(image_audit.columns):\n    manifest_clean = manifest_clean.merge(\n        image_audit[\n            [\n                \"image_id\", \"export_sha256\",\n                \"export_pixel_sha256\",\n            ]\n        ],\n        on=\"image_id\",\n        how=\"left\",\n        validate=\"one_to_one\",\n    )\n    manifest_clean.to_csv(\n        DATASET_ROOT / \"manifest.csv\", index=False\n    )\n\nfull_image_audit_completed = (\n    AUDIT_MODE == \"full\"\n    and len(image_audit) == len(manifest_clean)\n    and image_audit[\"image_id\"].nunique() == len(manifest_clean)\n    and set(image_audit[\"image_id\"])\n    == set(manifest_clean[\"image_id\"])\n)\nsource_baseline_hash_available = bool(\n    image_audit[\"source_baseline_available\"].any()\n)\nbaseline_hash_mismatches = image_audit.loc[\n    image_audit[\"source_baseline_available\"]\n    & ~image_audit[\"source_baseline_hash_matches\"].eq(True)\n].copy()\nexport_hash_mismatches = image_audit.loc[\n    ~image_audit[\"export_hash_matches_source\"].eq(True)\n].copy()\nexport_pixel_hash_mismatches = image_audit.loc[\n    ~image_audit[\"export_pixel_hash_matches_source\"].eq(True)\n].copy()\n\nduplicate_content_rows = pd.DataFrame()\ncontent_hash_split_leakage = pd.DataFrame()\nsame_split_duplicate_content = pd.DataFrame()\nif (\n    full_image_audit_completed\n    and \"export_pixel_sha256\" in image_audit.columns\n):\n    pixel_group_stats = (\n        image_audit.groupby(\"export_pixel_sha256\")\n        .agg(\n            image_count=(\"image_id\", \"nunique\"),\n            split_count=(\"split\", \"nunique\"),\n        )\n        .reset_index()\n    )\n    duplicate_hashes = set(\n        pixel_group_stats.loc[\n            pixel_group_stats[\"image_count\"] > 1,\n            \"export_pixel_sha256\",\n        ]\n    )\n    cross_split_hashes = set(\n        pixel_group_stats.loc[\n            (pixel_group_stats[\"image_count\"] > 1)\n            & (pixel_group_stats[\"split_count\"] > 1),\n            \"export_pixel_sha256\",\n        ]\n    )\n    same_split_hashes = duplicate_hashes - cross_split_hashes\n    duplicate_content_rows = image_audit.loc[\n        image_audit[\"export_pixel_sha256\"].isin(\n            duplicate_hashes\n        )\n    ].sort_values(\n        [\"export_pixel_sha256\", \"split\", \"image_id\"]\n    )\n    content_hash_split_leakage = image_audit.loc[\n        image_audit[\"export_pixel_sha256\"].isin(\n            cross_split_hashes\n        )\n    ].sort_values(\n        [\"export_pixel_sha256\", \"split\", \"image_id\"]\n    )\n    same_split_duplicate_content = image_audit.loc[\n        image_audit[\"export_pixel_sha256\"].isin(\n            same_split_hashes\n        )\n    ].sort_values(\n        [\"export_pixel_sha256\", \"split\", \"image_id\"]\n    )\n\nduplicate_content_rows.to_csv(\n    REPORT_ROOT / \"duplicate_image_content.csv\", index=False\n)\ncontent_hash_split_leakage.to_csv(\n    REPORT_ROOT / \"content_hash_split_leakage.csv\", index=False\n)\nsame_split_duplicate_content.to_csv(\n    REPORT_ROOT / \"same_split_duplicate_content_review.csv\",\n    index=False,\n)\n\nprint(\"Audit mode:\", AUDIT_MODE)\nprint(\"Audited PNGs:\", len(image_audit), \"/\", len(manifest_clean))\nprint(\"Failed image audit:\", len(failed_image_audit))\nprint(\"Non-blocking image-quality review rows:\",\n      len(image_quality_review))\nprint(\"Full image audit completed:\", full_image_audit_completed)\nprint(\"Source preprocessing hash baseline available:\",\n      source_baseline_hash_available)\nprint(\"Export file-hash mismatches:\",\n      len(export_hash_mismatches))\nprint(\"Export pixel-hash mismatches:\",\n      len(export_pixel_hash_mismatches))\nprint(\"Residual duplicate-pixel rows:\",\n      len(duplicate_content_rows))\nprint(\"Cross-split duplicate-pixel rows:\",\n      len(content_hash_split_leakage))\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 11. 校验 split plan、分组泄漏和可用的患者级信息\nsplit_plan_all = pd.read_csv(source_split_plan_path)\nrequired_split_plan_columns = {\"image_id\", \"split\", \"group_id\"}\nmissing_split_plan_columns = (\n    required_split_plan_columns - set(split_plan_all.columns)\n)\nif missing_split_plan_columns:\n    raise ValueError(\n        f\"split_plan.csv 缺少字段：\"\n        f\"{sorted(missing_split_plan_columns)}\"\n    )\n\nsplit_plan_all[\"image_id\"] = (\n    split_plan_all[\"image_id\"].astype(str).str.strip()\n)\nsplit_plan_all[\"split\"] = (\n    split_plan_all[\"split\"].astype(str).str.strip().str.lower()\n)\nsplit_plan_all[\"group_id\"] = (\n    split_plan_all[\"group_id\"].astype(str).str.strip()\n)\ninvalid_id_tokens = {\"\", \"nan\", \"none\", \"null\"}\ninvalid_split_plan_image_ids = split_plan_all.loc[\n    split_plan_all[\"image_id\"].str.lower().isin(\n        invalid_id_tokens\n    )\n].copy()\nduplicate_split_plan_rows = split_plan_all.loc[\n    split_plan_all.duplicated(\"image_id\", keep=False)\n].copy()\nsplit_plan_unique = split_plan_all.loc[\n    ~split_plan_all.duplicated(\"image_id\", keep=False)\n].copy()\ninvalid_split_plan_splits = split_plan_all.loc[\n    ~split_plan_all[\"split\"].isin(VALID_SPLITS)\n].copy()\ninvalid_group_rows_all = split_plan_all.loc[\n    split_plan_all[\"group_id\"].str.lower().isin(\n        invalid_id_tokens\n    )\n].copy()\n\nsource_manifest_id_set = set(manifest[\"image_id\"])\nsource_split_plan_id_set = set(split_plan_all[\"image_id\"])\nmissing_source_split_plan_ids = sorted(\n    source_manifest_id_set - source_split_plan_id_set\n)\nextra_split_plan_ids = sorted(\n    source_split_plan_id_set - source_manifest_id_set\n)\nsource_split_plan_comparison = manifest[\n    [\"image_id\", \"split\"]\n].merge(\n    split_plan_unique[[\"image_id\", \"split\"]].rename(\n        columns={\"split\": \"planned_split\"}\n    ),\n    on=\"image_id\",\n    how=\"left\",\n    validate=\"one_to_one\",\n)\nsource_split_plan_mismatch = source_split_plan_comparison.loc[\n    source_split_plan_comparison[\"split\"]\n    != source_split_plan_comparison[\"planned_split\"]\n].copy()\n\nsplit_plan = split_plan_unique.loc[\n    split_plan_unique[\"image_id\"].isin(manifest_clean[\"image_id\"])\n].copy()\nmissing_split_plan_ids = sorted(\n    set(manifest_clean[\"image_id\"]) - set(split_plan[\"image_id\"])\n)\nsplit_plan_comparison = manifest_clean[\n    [\"image_id\", \"split\"]\n].merge(\n    split_plan[[\"image_id\", \"split\"]].rename(\n        columns={\"split\": \"planned_split\"}\n    ),\n    on=\"image_id\",\n    how=\"left\",\n    validate=\"one_to_one\",\n)\nsplit_plan_mismatch = split_plan_comparison.loc[\n    split_plan_comparison[\"split\"]\n    != split_plan_comparison[\"planned_split\"]\n].copy()\n\ngroup_split_counts = (\n    split_plan.groupby(\"group_id\")[\"split\"]\n    .agg(\n        split_count=\"nunique\",\n        splits=lambda values: \"|\".join(\n            sorted(set(values))\n        ),\n        image_count=\"size\",\n    )\n    .reset_index()\n)\ngroup_split_leakage = group_split_counts.loc[\n    group_split_counts[\"split_count\"] > 1\n].copy()\ngroup_size_distribution = (\n    split_plan.groupby(\"group_id\").size()\n    .value_counts()\n    .sort_index()\n    .rename_axis(\"images_per_group\")\n    .reset_index(name=\"group_count\")\n)\ngroup_id_equals_image_id_fraction = float(\n    (\n        split_plan[\"group_id\"].astype(str)\n        == split_plan[\"image_id\"].astype(str)\n    ).mean()\n) if len(split_plan) else 0.0\ngroup_id_semantics = (\n    source_config.get(\"split_group_semantics\")\n    or source_config.get(\"group_id_source\")\n    or \"source_split_plan_group_id\"\n)\n\npatient_id_column = next(\n    (\n        column for column in (\n            \"patient_id\", \"PatientID\", \"patient_hash\"\n        )\n        if column in split_plan.columns\n    ),\n    None,\n)\npatient_split_leakage = pd.DataFrame()\nif patient_id_column is not None:\n    patient_values = (\n        split_plan[patient_id_column].astype(str).str.strip()\n    )\n    invalid_patient_rows = split_plan.loc[\n        patient_values.str.lower().isin(invalid_id_tokens)\n    ].copy()\n    patient_split_counts = (\n        split_plan.assign(_patient_id=patient_values)\n        .groupby(\"_patient_id\")[\"split\"]\n        .agg(\n            split_count=\"nunique\",\n            splits=lambda values: \"|\".join(\n                sorted(set(values))\n            ),\n        )\n        .reset_index()\n    )\n    patient_split_leakage = patient_split_counts.loc[\n        patient_split_counts[\"split_count\"] > 1\n    ].copy()\nelse:\n    invalid_patient_rows = pd.DataFrame()\n\ninvalid_split_plan_image_ids.to_csv(\n    REPORT_ROOT / \"invalid_split_plan_image_ids.csv\",\n    index=False,\n)\nduplicate_split_plan_rows.to_csv(\n    REPORT_ROOT / \"duplicate_split_plan_rows.csv\", index=False\n)\ninvalid_split_plan_splits.to_csv(\n    REPORT_ROOT / \"invalid_split_plan_splits.csv\", index=False\n)\ninvalid_group_rows_all.to_csv(\n    REPORT_ROOT / \"invalid_split_group_ids.csv\", index=False\n)\npd.DataFrame({\n    \"image_id\": missing_source_split_plan_ids\n}).to_csv(\n    REPORT_ROOT / \"missing_source_split_plan_ids.csv\",\n    index=False,\n)\npd.DataFrame({\n    \"image_id\": extra_split_plan_ids\n}).to_csv(\n    REPORT_ROOT / \"extra_split_plan_ids.csv\", index=False\n)\nsource_split_plan_mismatch.to_csv(\n    REPORT_ROOT / \"source_split_plan_mismatch.csv\",\n    index=False,\n)\npd.DataFrame({\n    \"image_id\": missing_split_plan_ids\n}).to_csv(\n    REPORT_ROOT / \"missing_clean_split_plan_ids.csv\",\n    index=False,\n)\nsplit_plan_mismatch.to_csv(\n    REPORT_ROOT / \"clean_split_plan_mismatch.csv\",\n    index=False,\n)\ngroup_split_leakage.to_csv(\n    REPORT_ROOT / \"group_split_leakage.csv\", index=False\n)\ngroup_size_distribution.to_csv(\n    REPORT_ROOT / \"group_size_distribution.csv\", index=False\n)\npatient_split_leakage.to_csv(\n    REPORT_ROOT / \"patient_split_leakage.csv\", index=False\n)\ninvalid_patient_rows.to_csv(\n    REPORT_ROOT / \"invalid_patient_ids.csv\", index=False\n)\n(REPORT_ROOT / \"group_id_semantics.json\").write_text(\n    json.dumps({\n        \"group_id_semantics\": group_id_semantics,\n        \"patient_id_column_available\": patient_id_column,\n        \"group_id_equals_image_id_fraction\": (\n            group_id_equals_image_id_fraction\n        ),\n        \"group_count\": int(\n            split_plan[\"group_id\"].nunique()\n        ),\n    }, ensure_ascii=False, indent=2),\n    encoding=\"utf-8\",\n)\n\nclean_split_plan_path = DATASET_ROOT / \"split_plan.csv\"\nsplit_plan = split_plan.sort_values(\n    [\"split\", \"image_id\"], kind=\"stable\"\n).reset_index(drop=True)\nsplit_plan.to_csv(clean_split_plan_path, index=False)\nsplit_plan.to_csv(\n    REPORT_ROOT / \"clean_split_plan.csv\", index=False\n)\nclean_split_plan_written = (\n    clean_split_plan_path.is_file()\n    and len(split_plan) == len(manifest_clean)\n    and split_plan[\"image_id\"].nunique() == len(manifest_clean)\n    and set(split_plan[\"image_id\"])\n    == set(manifest_clean[\"image_id\"])\n)\n\nprint(\"Image-ID split leakage:\", len(image_split_leakage))\nprint(\"Group-level split leakage:\", len(group_split_leakage))\nprint(\"Patient ID column available:\", patient_id_column)\nprint(\"Patient-level split leakage rows:\",\n      len(patient_split_leakage))\nprint(\"Missing source/clean split-plan IDs:\",\n      len(missing_source_split_plan_ids),\n      len(missing_split_plan_ids))\nprint(\"Duplicate split-plan rows:\",\n      len(duplicate_split_plan_rows))\nprint(\"Source/clean split-plan mismatches:\",\n      len(source_split_plan_mismatch),\n      len(split_plan_mismatch))\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 12. 最终训练闸门与行动清单\n\n`TRAINING_READY` 要求来源锁、完整源图审计、像素级重复审计、序列化几何、\n标签重建、全量导出审计、split/group 泄漏和数据指纹全部通过。单医生框、\n低对比度图和感知哈希疑似近重复会保留在复核报告中；像素相同但标签冲突\n属于硬性阻断项。\n","metadata":{}},{"cell_type":"code","source":"changed_label_images = int(\n    (~label_comparison[\"same_labels_rounded_6dp\"]).sum()\n)\nretained_clipped_boxes = int(\n    box_issues[\"action\"].eq(\"clipped_for_review\").sum()\n)\ninvalid_auto_excluded_images_retained = (\n    set(automatic_source_image_exclusions[\"image_id\"])\n    & set(manifest_clean[\"image_id\"])\n)\n\nacceptance_checks = {\n    \"source_lock_verified\": (\n        source_output_root.name\n        == EXPECTED_SOURCE_OUTPUT_DIRNAME\n        and signature_matches\n        and computed_signature\n        == EXPECTED_SOURCE_DATASET_SIGNATURE\n        and (\n            source_notebook_hint_matches\n            or not REQUIRE_NOTEBOOK_NAME_HINT\n        )\n        and source_config.get(\"target_class\") == TARGET_CLASS\n    ),\n    \"raw_dicom_read_is_false\": True,\n    \"raw_train_csv_read_is_false\": True,\n    \"no_dicom_inside_selected_dataset\": (\n        len(dicom_inside_selected_dataset) == 0\n    ),\n    \"source_count_contract_passed\": source_count_contract_pass,\n    \"source_yaml_class_matches\": source_yaml_class_matches,\n    \"source_manifest_box_contract_passed\": (\n        source_manifest_box_contract_pass\n    ),\n    \"geometry_manifest_contract_passed\": (\n        geometry_manifest_contract_pass\n    ),\n    \"algebraic_geometry_check_passed\": geometry_roundtrip_pass,\n    \"serialized_yolo_geometry_passed\": serialized_geometry_pass,\n    \"radiologist_metadata_valid\": radiologist_metadata_valid,\n    \"no_duplicate_manifest_image_ids\": (\n        len(duplicate_manifest_rows) == 0\n    ),\n    \"valid_manifest_image_ids\": (\n        len(invalid_manifest_image_ids) == 0\n    ),\n    \"valid_manifest_status_and_dimensions\": (\n        len(invalid_status_rows) == 0\n        and len(invalid_dimension_rows) == 0\n    ),\n    \"no_missing_preprocessed_png_or_label\": (\n        len(missing_source_files) == 0\n    ),\n    \"no_extra_preprocessed_images\": (\n        len(extra_preprocessed_images) == 0\n    ),\n    \"no_extra_preprocessed_labels\": (\n        len(extra_preprocessed_labels) == 0\n    ),\n    \"no_unexpected_source_files\": (\n        len(unexpected_source_files) == 0\n    ),\n    \"preexport_source_image_audit_complete\": (\n        preexport_source_audit_complete\n    ),\n    \"kept_source_file_and_pixel_hashes_complete\": (\n        kept_source_hashes_complete\n    ),\n    \"no_source_provenance_hash_failures\": (\n        len(source_provenance_failures) == 0\n    ),\n    \"automatic_invalid_image_exclusion_within_limit\": (\n        automatic_source_image_exclusion_within_limit\n    ),\n    \"no_auto_excluded_images_retained\": (\n        len(invalid_auto_excluded_images_retained) == 0\n    ),\n    \"exact_pixel_duplicate_audit_complete\": (\n        len(duplicate_candidates) == len(duplicate_decisions)\n    ),\n    \"no_unresolved_duplicate_label_conflicts\": (\n        no_duplicate_label_conflicts\n    ),\n    \"no_positive_images_missing_manifest\": (\n        len(positive_ids_missing_manifest) == 0\n    ),\n    \"source_positive_status_matches_rebuilt_boxes\": (\n        len(source_positive_mismatch) == 0\n    ),\n    \"positive_and_negative_images_present\": (\n        class_distribution_valid\n    ),\n    \"all_splits_have_positive_and_negative_images\": (\n        all_splits_have_both_classes\n    ),\n    \"ambiguous_positive_exclusion_within_limit\": (\n        ambiguous_positive_fraction\n        <= MAX_AMBIGUOUS_POSITIVE_FRACTION\n    ),\n    \"disk_preflight_passed\": disk_preflight_pass,\n    \"no_export_failures\": len(export_failures_df) == 0,\n    \"all_usable_images_exported\": (\n        len(manifest_clean) == len(usable_manifest)\n    ),\n    \"no_missing_output_images\": (\n        len(missing_output_images) == 0\n    ),\n    \"no_missing_output_labels\": (\n        len(missing_output_labels) == 0\n    ),\n    \"no_extra_output_images\": len(extra_output_images) == 0,\n    \"no_extra_output_labels\": len(extra_output_labels) == 0,\n    \"no_unexpected_output_directories\": (\n        len(unexpected_output_directories_df) == 0\n    ),\n    \"no_unexpected_output_entries\": (\n        len(postexport_unexpected_entries_df) == 0\n    ),\n    \"portable_data_yaml_resolves\": (\n        IMAGE_EXPORT_MODE == \"none\" or yaml_paths_exist\n    ),\n    \"portable_data_yaml_class_matches\": (\n        IMAGE_EXPORT_MODE == \"none\" or yaml_class_matches\n    ),\n    \"no_invalid_rebuilt_labels\": len(clean_label_errors) == 0,\n    \"rebuilt_label_counts_match_manifest\": (\n        len(clean_label_count_mismatch) == 0\n    ),\n    \"full_image_audit_completed\": full_image_audit_completed,\n    \"image_audit_passed\": len(failed_image_audit) == 0,\n    \"export_file_hashes_match_preexport_source\": (\n        len(export_hash_mismatches) == 0\n    ),\n    \"export_pixel_hashes_match_preexport_source\": (\n        len(export_pixel_hash_mismatches) == 0\n    ),\n    \"available_source_baseline_hashes_match\": (\n        len(baseline_hash_mismatches) == 0\n    ),\n    \"no_image_id_split_leakage\": len(image_split_leakage) == 0,\n    \"no_group_level_split_leakage\": (\n        len(group_split_leakage) == 0\n    ),\n    \"available_patient_ids_do_not_leak\": (\n        patient_id_column is None\n        or (\n            len(invalid_patient_rows) == 0\n            and len(patient_split_leakage) == 0\n        )\n    ),\n    \"source_split_plan_complete\": (\n        len(missing_source_split_plan_ids) == 0\n    ),\n    \"no_extra_source_split_plan_ids\": (\n        len(extra_split_plan_ids) == 0\n    ),\n    \"split_plan_image_ids_unique\": (\n        len(duplicate_split_plan_rows) == 0\n    ),\n    \"split_plan_image_ids_valid\": (\n        len(invalid_split_plan_image_ids) == 0\n    ),\n    \"split_plan_splits_valid\": (\n        len(invalid_split_plan_splits) == 0\n    ),\n    \"split_group_ids_valid\": (\n        len(invalid_group_rows_all) == 0\n    ),\n    \"source_split_plan_matches_manifest\": (\n        len(source_split_plan_mismatch) == 0\n    ),\n    \"clean_split_plan_complete\": (\n        len(missing_split_plan_ids) == 0\n    ),\n    \"clean_split_plan_matches_manifest\": (\n        len(split_plan_mismatch) == 0\n    ),\n    \"clean_split_plan_written\": clean_split_plan_written,\n    \"duplicate_pixel_report_written\": (\n        (REPORT_ROOT / \"duplicate_image_content.csv\").exists()\n    ),\n    \"cross_split_duplicate_report_written\": (\n        (REPORT_ROOT / \"content_hash_split_leakage.csv\").exists()\n    ),\n    \"dataset_fingerprint_created\": (\n        isinstance(CLEAN_DATASET_FINGERPRINT, str)\n        and len(CLEAN_DATASET_FINGERPRINT) == 64\n    ),\n}\nacceptance_checks = {\n    name: bool(passed)\n    for name, passed in acceptance_checks.items()\n}\n\nall_checks_pass = all(acceptance_checks.values())\nchecks_except_full_audit = {\n    name: passed\n    for name, passed in acceptance_checks.items()\n    if name != \"full_image_audit_completed\"\n}\nif all_checks_pass and IMAGE_EXPORT_MODE == \"copy\":\n    final_status = \"TRAINING_READY\"\nelif all_checks_pass and IMAGE_EXPORT_MODE == \"symlink\":\n    final_status = \"SESSION_ONLY\"\nelif all_checks_pass and IMAGE_EXPORT_MODE == \"none\":\n    final_status = \"REPORT_ONLY\"\nelif (\n    AUDIT_MODE == \"sample\"\n    and all(checks_except_full_audit.values())\n):\n    final_status = \"SAMPLE_ONLY\"\nelse:\n    final_status = \"FAIL\"\n\ntraining_ready = final_status == \"TRAINING_READY\"\nmanual_review_required = bool(\n    len(manual_review_queue) > 0\n    or retained_clipped_boxes > 0\n    or len(low_consensus_boxes) > 0\n    or len(image_quality_review) > 0\n    or len(cross_split_perceptual_hash_review) > 0\n)\n\naction_for_check = {\n    \"source_count_contract_passed\": (\n        \"确认添加的是完整 b68b716ab287 输出；\"\n        \"源清单应为 15000 张且 split 为 12000/1500/1500。\"\n    ),\n    \"source_yaml_class_matches\": (\n        \"源 data.yaml 不是单类 Nodule/Mass，禁止混用。\"\n    ),\n    \"source_manifest_box_contract_passed\": (\n        \"源 manifest 的 label_count/is_positive 与融合框数量不一致。\"\n    ),\n    \"geometry_manifest_contract_passed\": (\n        \"确认输入确实来自 640version1，且未做未记录的裁剪/补边。\"\n    ),\n    \"serialized_yolo_geometry_passed\": (\n        \"停止训练，核对标签小数序列化、坐标顺序和 PNG 尺寸。\"\n    ),\n    \"radiologist_metadata_valid\": (\n        \"radiologist_count 应是 1–3 的整数；核对融合框报告。\"\n    ),\n    \"preexport_source_image_audit_complete\": (\n        \"源图未完成全量解码审计，重新 Run All。\"\n    ),\n    \"no_source_provenance_hash_failures\": (\n        \"源 PNG 与已有预处理哈希不一致，重新保存可信的 640version1。\"\n    ),\n    \"automatic_invalid_image_exclusion_within_limit\": (\n        \"坏图比例过高，说明预处理可能整体异常；\"\n        \"查看 automatic_source_image_exclusions.csv。\"\n    ),\n    \"no_unresolved_duplicate_label_conflicts\": (\n        \"查看 duplicate_label_conflicts_quarantined.csv；\"\n        \"同像素不同标签必须由合格人员解决后再训练。\"\n    ),\n    \"all_splits_have_positive_and_negative_images\": (\n        \"回到 640version1 重新划分，保证三份 split 都有阳性和阴性。\"\n    ),\n    \"ambiguous_positive_exclusion_within_limit\": (\n        \"异常阳性排除比例过高，先复核或重新标框。\"\n    ),\n    \"disk_preflight_passed\": (\n        \"清理 /kaggle/working；正式保存仍使用 copy。\"\n    ),\n    \"no_export_failures\": (\n        \"查看 export_failures.csv，修复后重新 Run All。\"\n    ),\n    \"full_image_audit_completed\": (\n        \"把 AUDIT_MODE 改为 full 并 Run All；抽样不能正式训练。\"\n    ),\n    \"image_audit_passed\": (\n        \"查看 failed_image_audit.csv；最终复制件未通过解码或尺寸检查。\"\n    ),\n    \"export_file_hashes_match_preexport_source\": (\n        \"导出字节与预导出源图不同，删除对应 V6 复制件后重跑。\"\n    ),\n    \"export_pixel_hashes_match_preexport_source\": (\n        \"导出解码像素与源图不同，禁止训练。\"\n    ),\n    \"no_group_level_split_leakage\": (\n        \"回到 640version1 按 source group_id 重新划分。\"\n    ),\n    \"available_patient_ids_do_not_leak\": (\n        \"发现患者 ID 跨 split；必须在预处理阶段重新分组划分。\"\n    ),\n    \"source_split_plan_complete\": (\n        \"修复 640version1 的 split_plan.csv；不能自行补猜分组。\"\n    ),\n    \"source_split_plan_matches_manifest\": (\n        \"源 manifest 与 split_plan 的 split 不一致。\"\n    ),\n    \"duplicate_pixel_report_written\": (\n        \"未生成重复像素报告，查看 duplicate_image_content.csv。\"\n    ),\n    \"cross_split_duplicate_report_written\": (\n        \"未生成跨 split 重复报告，查看 content_hash_split_leakage.csv。\"\n    ),\n    \"dataset_fingerprint_created\": (\n        \"未生成稳定数据指纹，禁止训练 Notebook 接入。\"\n    ),\n}\naction_rows = []\nfor check_name, passed in acceptance_checks.items():\n    if not passed:\n        action_rows.append({\n            \"priority\": \"BLOCKER\",\n            \"item\": check_name,\n            \"action\": action_for_check.get(\n                check_name,\n                \"查看对应 reports 文件，修复来源后重新 Run All。\",\n            ),\n        })\nif IMAGE_EXPORT_MODE != \"copy\":\n    action_rows.append({\n        \"priority\": \"BLOCKER\",\n        \"item\": \"non_portable_export\",\n        \"action\": \"正式保存/共享必须使用 IMAGE_EXPORT_MODE='copy'。\",\n    })\nif len(automatic_source_image_exclusions):\n    action_rows.append({\n        \"priority\": \"NOTE\",\n        \"item\": \"invalid_source_images_auto_excluded\",\n        \"action\": (\n            f\"V6 已自动隔离 {len(automatic_source_image_exclusions)} \"\n            \"张损坏或违反 PNG 契约的图片；源文件未修改。\"\n        ),\n    })\nif len(auto_duplicate_exclusions):\n    action_rows.append({\n        \"priority\": \"NOTE\",\n        \"item\": \"exact_pixel_duplicates_auto_excluded\",\n        \"action\": (\n            f\"V6 已自动排除 {len(auto_duplicate_exclusions)} \"\n            \"个像素和标签均相同的副本；源文件未修改。\"\n        ),\n    })\nif len(retained_exact_pixel_duplicates):\n    action_rows.append({\n        \"priority\": \"NOTE\",\n        \"item\": \"exact_pixel_duplicates_retained_report_only\",\n        \"action\": (\n            f\"已按复现实验策略保留 {len(retained_exact_pixel_duplicates)} \"\n            \"个像素和标签均相同的额外副本；详见 \"\n            \"retained_exact_pixel_duplicates.csv。\"\n        ),\n    })\nif len(content_hash_split_leakage):\n    action_rows.append({\n        \"priority\": \"REVIEW\",\n        \"item\": \"cross_split_exact_duplicates_retained\",\n        \"action\": (\n            \"已关闭自动去重并保留跨 split 的相同像素图；\"\n            \"论文中应披露潜在数据泄漏风险，详见 \"\n            \"content_hash_split_leakage.csv。\"\n        ),\n    })\nif len(duplicate_label_conflicts):\n    action_rows.append({\n        \"priority\": \"BLOCKER\",\n        \"item\": \"duplicate_label_conflicts_quarantined\",\n        \"action\": (\n            \"冲突组已全部隔离，但 V6 会保持 FAIL，\"\n            \"直到合格人员解决标签冲突。\"\n        ),\n    })\nif len(manual_review_queue):\n    action_rows.append({\n        \"priority\": \"REVIEW\",\n        \"item\": \"manual_review_queue\",\n        \"action\": (\n            \"查看 manual_review_queue.csv；被隔离的阳性图\"\n            \"不能自动改成负样本。\"\n        ),\n    })\nif retained_clipped_boxes:\n    action_rows.append({\n        \"priority\": \"REVIEW\",\n        \"item\": \"retained_clipped_boxes\",\n        \"action\": \"检查 clipped_box_review.csv 和最终框可视化。\",\n    })\nif len(low_consensus_boxes):\n    action_rows.append({\n        \"priority\": \"REVIEW\",\n        \"item\": \"single_radiologist_boxes\",\n        \"action\": (\n            \"这是敏感性优先策略保留的低共识框；\"\n            \"报告时不要当作多医生共识真值。\"\n        ),\n    })\nif len(image_quality_review):\n    action_rows.append({\n        \"priority\": \"REVIEW\",\n        \"item\": \"image_quality_outliers\",\n        \"action\": (\n            \"查看 image_quality_review.csv；\"\n            \"亮度统计异常只复核，不武断删除医学图像。\"\n        ),\n    })\nif len(cross_split_perceptual_hash_review):\n    action_rows.append({\n        \"priority\": \"REVIEW\",\n        \"item\": \"perceptual_near_duplicate_candidates\",\n        \"action\": (\n            \"查看 cross_split_perceptual_hash_review.csv；\"\n            \"感知哈希仅作疑似近重复提示，不自动删除。\"\n        ),\n    })\nif changed_label_images:\n    action_rows.append({\n        \"priority\": \"REQUIRED\",\n        \"item\": \"labels_changed\",\n        \"action\": (\n            \"正式实验必须使用 V6 data.yaml 从新权重重新训练；\"\n            \"旧 best.pt 不代表修正后的标签。\"\n        ),\n    })\nif not source_baseline_hash_available:\n    action_rows.append({\n        \"priority\": \"NOTE\",\n        \"item\": \"no_preprocessing_hash_baseline\",\n        \"action\": (\n            \"V6 已为当前源输出建立完整指纹；\"\n            \"下次预处理应把 image_sha256 直接写入 manifest。\"\n        ),\n    })\nif patient_id_column is None:\n    action_rows.append({\n        \"priority\": \"NOTE\",\n        \"item\": \"patient_id_not_explicitly_available\",\n        \"action\": (\n            \"V6 已检查 source group_id；报告中不要把它\"\n            \"无条件写成患者 ID，除非 640version1 明确记录其来源。\"\n        ),\n    })\nif training_ready:\n    action_rows.append({\n        \"priority\": \"NEXT\",\n        \"item\": \"start_training\",\n        \"action\": (\n            \"Save Version；训练 Notebook 读取 dataset/data.yaml，\"\n            \"并核对 CLEAN_DATASET_FINGERPRINT 后从新权重训练。\"\n        ),\n    })\naction_plan = pd.DataFrame(\n    action_rows,\n    columns=[\"priority\", \"item\", \"action\"],\n)\naction_plan.to_csv(\n    REPORT_ROOT / \"action_plan.csv\", index=False\n)\n\nsummary = pd.DataFrame([\n    {\"metric\": \"Final status\", \"value\": final_status},\n    {\"metric\": \"Training ready\", \"value\": training_ready},\n    {\"metric\": \"Input policy\",\n     \"value\": \"640version1 preprocessed output only\"},\n    {\"metric\": \"Source configuration signature\",\n     \"value\": computed_signature},\n    {\"metric\": \"Cleaning policy signature\",\n     \"value\": CLEANING_POLICY_SIGNATURE},\n    {\"metric\": \"Source dataset fingerprint\",\n     \"value\": SOURCE_DATASET_FINGERPRINT},\n    {\"metric\": \"Clean dataset fingerprint\",\n     \"value\": CLEAN_DATASET_FINGERPRINT},\n    {\"metric\": \"Source manifest images\", \"value\": len(manifest)},\n    {\"metric\": \"Automatically excluded invalid images\",\n     \"value\": len(automatic_source_image_exclusions)},\n    {\"metric\": \"Usable clean images\", \"value\": len(manifest_clean)},\n    {\"metric\": \"Duplicate decoded-pixel groups\",\n     \"value\": deduplication_summary[\"duplicate_content_groups\"]},\n    {\"metric\": \"Exact duplicate copies retained (report only)\",\n     \"value\": len(retained_exact_pixel_duplicates)},\n    {\"metric\": \"Exact duplicate copies auto-excluded\",\n     \"value\": len(auto_duplicate_exclusions)},\n    {\"metric\": \"Duplicate label-conflict images quarantined\",\n     \"value\": len(duplicate_label_conflicts)},\n    {\"metric\": \"Source fused boxes\", \"value\": len(source_fused_boxes)},\n    {\"metric\": \"Clean fused boxes\", \"value\": len(clean_fused_boxes)},\n    {\"metric\": \"Excluded ambiguous positive images\",\n     \"value\": len(ambiguous_ids)},\n    {\"metric\": \"Maximum serialized geometry error (px)\",\n     \"value\": max_serialized_geometry_error},\n    {\"metric\": \"Single-radiologist boxes for review\",\n     \"value\": len(low_consensus_boxes)},\n    {\"metric\": \"Labels changed by cleaning\",\n     \"value\": changed_label_images},\n    {\"metric\": \"Audited final images\", \"value\": len(image_audit)},\n    {\"metric\": \"Failed final image audits\",\n     \"value\": len(failed_image_audit)},\n    {\"metric\": \"Cross-split perceptual review rows\",\n     \"value\": len(cross_split_perceptual_hash_review)},\n    {\"metric\": \"Manual review required\",\n     \"value\": manual_review_required},\n])\nsummary.to_csv(\n    REPORT_ROOT / \"cleaning_summary.csv\", index=False\n)\ndisplay(summary)\ndisplay(action_plan)\n\nRUN_COMPLETED_AT = datetime.now(timezone.utc).isoformat()\nacceptance_report = {\n    \"overall_status\": final_status,\n    \"training_ready\": bool(training_ready),\n    \"manual_review_required\": bool(manual_review_required),\n    \"input_policy\": \"640version1_preprocessed_output_only\",\n    \"cleaning_version\": \"6.0-release\",\n    \"gpu_used\": False,\n    \"raw_dicom_read\": False,\n    \"raw_train_csv_read\": False,\n    \"run_started_at\": RUN_STARTED_AT,\n    \"run_completed_at\": RUN_COMPLETED_AT,\n    \"source_dataset_root\": str(source_dataset_root),\n    \"source_dataset_signature\": computed_signature,\n    \"source_dataset_fingerprint\": SOURCE_DATASET_FINGERPRINT,\n    \"cleaning_policy_signature\": CLEANING_POLICY_SIGNATURE,\n    \"clean_dataset_fingerprint\": CLEAN_DATASET_FINGERPRINT,\n    \"run_root\": str(RUN_ROOT),\n    \"dataset_root\": str(DATASET_ROOT),\n    \"data_yaml\": (\n        str(data_yaml_path) if data_yaml_path.exists() else None\n    ),\n    \"checks\": acceptance_checks,\n    \"counts\": {\n        \"source_manifest_images\": int(len(manifest)),\n        \"automatically_excluded_invalid_images\": int(\n            len(automatic_source_image_exclusions)\n        ),\n        \"usable_clean_images\": int(len(manifest_clean)),\n        \"duplicate_content_groups\": int(\n            deduplication_summary[\"duplicate_content_groups\"]\n        ),\n        \"safe_duplicate_copies_auto_excluded\": int(\n            len(auto_duplicate_exclusions)\n        ),\n        \"exact_duplicate_copies_retained_report_only\": int(\n            len(retained_exact_pixel_duplicates)\n        ),\n        \"duplicate_label_conflict_images_quarantined\": int(\n            len(duplicate_label_conflicts)\n        ),\n        \"source_fused_boxes\": int(len(source_fused_boxes)),\n        \"clean_fused_boxes\": int(len(clean_fused_boxes)),\n        \"ambiguous_positive_images_excluded\": int(\n            len(ambiguous_ids)\n        ),\n        \"single_radiologist_boxes_for_review\": int(\n            len(low_consensus_boxes)\n        ),\n        \"changed_label_images\": int(changed_label_images),\n        \"failed_image_audits\": int(len(failed_image_audit)),\n        \"image_quality_review_rows\": int(\n            len(image_quality_review)\n        ),\n        \"cross_split_perceptual_hash_review_rows\": int(\n            len(cross_split_perceptual_hash_review)\n        ),\n    },\n}\n(REPORT_ROOT / \"cleaning_acceptance_final.json\").write_text(\n    json.dumps(\n        acceptance_report, ensure_ascii=False, indent=2\n    ),\n    encoding=\"utf-8\",\n)\n(REPORT_ROOT / \"TRAINING_GATE.txt\").write_text(\n    (\n        f\"STATUS={final_status}\\n\"\n        f\"TRAINING_READY={training_ready}\\n\"\n        f\"MANUAL_REVIEW_REQUIRED={manual_review_required}\\n\"\n        f\"CLEANING_VERSION=6.0-release\\n\"\n        f\"SOURCE_DATASET_SIGNATURE={computed_signature}\\n\"\n        f\"SOURCE_DATASET_FINGERPRINT=\"\n        f\"{SOURCE_DATASET_FINGERPRINT}\\n\"\n        f\"CLEAN_DATASET_FINGERPRINT=\"\n        f\"{CLEAN_DATASET_FINGERPRINT}\\n\"\n        f\"DATA_YAML=\"\n        f\"{data_yaml_path if data_yaml_path.exists() else ''}\\n\"\n        f\"RUN_STARTED_AT={RUN_STARTED_AT}\\n\"\n        f\"RUN_COMPLETED_AT={RUN_COMPLETED_AT}\\n\"\n    ),\n    encoding=\"utf-8\",\n)\n\nnext_steps_lines = [\n    \"# V6 next steps\",\n    \"\",\n    f\"- Final status: `{final_status}`\",\n    f\"- Training ready: `{training_ready}`\",\n    f\"- Manual review required: `{manual_review_required}`\",\n    f\"- Clean dataset fingerprint: \"\n    f\"`{CLEAN_DATASET_FINGERPRINT}`\",\n    \"\",\n]\nfor row in action_plan.itertuples(index=False):\n    next_steps_lines.append(\n        f\"- [{row.priority}] `{row.item}`: {row.action}\"\n    )\n(RUN_ROOT / \"README_NEXT_STEPS.md\").write_text(\n    \"\\n\".join(next_steps_lines) + \"\\n\",\n    encoding=\"utf-8\",\n)\n\nprint(json.dumps(\n    acceptance_report, ensure_ascii=False, indent=2\n))\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 13. 在实际预处理 PNG 上可视化最终重建标签\npositive_preview = manifest_clean.loc[\n    manifest_clean[\"is_positive\"].eq(1)\n].sample(\n    n=min(\n        PREVIEW_POSITIVE_IMAGES,\n        int(manifest_clean[\"is_positive\"].sum()),\n    ),\n    random_state=SEED,\n)\n\nif len(positive_preview):\n    columns = min(3, len(positive_preview))\n    rows = math.ceil(len(positive_preview) / columns)\n    figure, axes = plt.subplots(rows, columns, figsize=(6 * columns, 7 * rows))\n    axes = np.atleast_1d(axes).reshape(-1)\n    for axis, row in zip(axes, positive_preview.itertuples(index=False)):\n        with Image.open(row.image_path) as source:\n            image = source.convert(\"RGB\")\n        draw = ImageDraw.Draw(image)\n        boxes, _ = parse_label(Path(row.label_path))\n        for _, x_center, y_center, width, height in boxes:\n            x1 = (x_center - width / 2) * image.width\n            y1 = (y_center - height / 2) * image.height\n            x2 = (x_center + width / 2) * image.width\n            y2 = (y_center + height / 2) * image.height\n            draw.rectangle(\n                [x1, y1, x2, y2],\n                outline=\"red\",\n                width=max(2, image.width // 400),\n            )\n        axis.imshow(image)\n        axis.set_title(\n            f\"{row.split} | boxes={row.label_count}\\n\"\n            f\"{row.image_id[:20]}…\"\n        )\n        axis.axis(\"off\")\n    for axis in axes[len(positive_preview):]:\n        axis.axis(\"off\")\n    figure.suptitle(\"Final V6 YOLO labels on preprocessed PNGs\", fontsize=18)\n    figure.tight_layout()\n    preview_path = REPORT_ROOT / \"final_label_preview.png\"\n    figure.savefig(preview_path, dpi=160, bbox_inches=\"tight\")\n    plt.show()\n    plt.close(figure)\n    print(\"Preview:\", preview_path)\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 14. 最终结论：只有 TRAINING_READY 才允许进入正式训练\nprint(\"=\" * 92)\nprint(\"FINAL CLEANING STATUS:\",\n      acceptance_report[\"overall_status\"])\nprint(\"TRAINING READY:\",\n      acceptance_report[\"training_ready\"])\nprint(\"INPUT POLICY: 640version1 PREPROCESSED OUTPUT ONLY\")\nprint(\"Raw DICOM read: False\")\nprint(\"Raw train.csv read: False\")\nprint(\"Clean dataset fingerprint:\",\n      acceptance_report[\"clean_dataset_fingerprint\"])\nprint(\"Acceptance:\",\n      REPORT_ROOT / \"cleaning_acceptance_final.json\")\nprint(\"Action plan:\", REPORT_ROOT / \"action_plan.csv\")\nprint(\"Training gate:\", REPORT_ROOT / \"TRAINING_GATE.txt\")\nif data_yaml_path.exists():\n    print(\"Training YAML:\", data_yaml_path)\nprint(\"=\" * 92)\n\nif (\n    FAIL_ON_NOT_TRAINING_READY\n    and not acceptance_report[\"training_ready\"]\n):\n    failed_checks = [\n        name for name, passed in acceptance_checks.items()\n        if not passed\n    ]\n    raise RuntimeError(\n        \"最终训练闸门未通过，禁止开始正式训练。状态：\"\n        f\"{acceptance_report['overall_status']}。失败项：\"\n        + \", \".join(failed_checks)\n    )\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# 直接复用本 Notebook 已生成的最终 manifest，避免读取不存在的临时路径。\ndf = manifest_clean.copy()\n\ndef complexity(row):\n    if row[\"is_positive\"] == 0 or row[\"label_count\"] == 0:\n        return \"negative\"\n    if row[\"label_count\"] == 1:\n        return \"positive_simple\"\n    if row[\"label_count\"] == 2:\n        return \"positive_medium\"\n    return \"positive_complex\"\n\ndf[\"complexity\"] = df.apply(complexity, axis=1)\n\nsimple_df = df[[\"image_id\", \"split\", \"is_positive\", \"label_count\", \"complexity\"]]\n\nsimple_df.to_csv(\"/kaggle/working/manifest_for_excel.csv\", index=False, encoding=\"utf-8-sig\")\nsimple_df.to_excel(\"/kaggle/working/manifest_for_excel.xlsx\", index=False)\n\nprint(simple_df[\"complexity\"].value_counts())\nprint(\"Saved:\")\nprint(\"/kaggle/working/manifest_for_excel.csv\")\nprint(\"/kaggle/working/manifest_for_excel.xlsx\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T15:24:09.966367Z","iopub.execute_input":"2026-08-09T15:24:09.967088Z","iopub.status.idle":"2026-08-09T15:24:11.233647Z","shell.execute_reply.started":"2026-08-09T15:24:09.967045Z","shell.execute_reply":"2026-08-09T15:24:11.232226Z"}},"outputs":[],"execution_count":null}]}