{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":15586689,"datasetId":9972578,"databundleVersionId":16518878},{"sourceType":"datasetVersion","sourceId":15094918,"datasetId":9664561,"databundleVersionId":15979687},{"sourceType":"datasetVersion","sourceId":15588375,"datasetId":9973820,"databundleVersionId":16520657}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!curl https://rclone.org/install.sh | sudo bash\n","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2026-04-08T03:56:01.73465Z","iopub.execute_input":"2026-04-08T03:56:01.734958Z","iopub.status.idle":"2026-04-08T03:56:04.514722Z","shell.execute_reply.started":"2026-04-08T03:56:01.734933Z","shell.execute_reply":"2026-04-08T03:56:04.514015Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!mkdir -p /root/.config/rclone\n!cp /kaggle/input/datasets/benxelua/drive123/rclone.conf /root/.config/rclone/rclone.conf\n!rclone lsd rclone:","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T03:56:04.516246Z","iopub.execute_input":"2026-04-08T03:56:04.51662Z","iopub.status.idle":"2026-04-08T03:56:05.795665Z","shell.execute_reply.started":"2026-04-08T03:56:04.516568Z","shell.execute_reply":"2026-04-08T03:56:05.794721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport json\nimport time\nimport random\nimport shutil\nimport subprocess\nfrom pathlib import Path\nfrom concurrent.futures import ThreadPoolExecutor, as_completed\n\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_modality_lut, apply_voi_lut\n\n# =========================\n# PATH CONFIG\n# =========================\nDATA_ROOT = Path(\"/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection\")\nTRAIN_ANN_CSV = Path(\"/kaggle/input/datasets/benxelua/correct-label/annotations/annotations_train.csv\")\nVAL_ANN_CSV = Path(\"/kaggle/input/datasets/benxelua/correct-label/annotations/annotations_test.csv\")\nTRAIN_LABELS_CSV = Path(\"/kaggle/input/datasets/benxelua/correct-label/annotations/image_labels_train.csv\")\nVAL_LABELS_CSV = Path(\"/kaggle/input/datasets/benxelua/correct-label/annotations/image_labels_test.csv\")\n\n\nPREPROCESS_MODE = \"voi_lut\"   # chọn: \"raw_minmax\" hoặc \"rescale_minmax\"\nOUT_ROOT = Path(f\"out_vindr_mmdet_{PREPROCESS_MODE}\")\nIMG_DIR = OUT_ROOT / \"images\"\nANN_DIR = OUT_ROOT / \"annotations\"\n\nIMG_TRAIN_DIR = IMG_DIR / \"train\"\nIMG_VAL_DIR = IMG_DIR / \"val\"\n\n# Thư mục tạm để chứa file ZIP trước khi đẩy lên rclone\nZIP_DIR = Path(\"temp_zips\")\n\nfor d in [IMG_TRAIN_DIR, IMG_VAL_DIR, ANN_DIR, ZIP_DIR]:\n    d.mkdir(parents=True, exist_ok=True)\n\n# =========================\n# SCRIPT CONFIG\n# =========================\nSEED = 42\nIMG_SIZE = 512\nSAVE_AS_JPG = False\nJPG_QUALITY = 95\nSTACK_CHANNELS_3 = True\n\n\n\n# =========================\n# RCLONE CONFIG\n# =========================\nBATCH_SIZE = 3000\n# Thư mục đích trên Google Drive\nRCLONE_DEST = f\"rclone:vinbigdata-{PREPROCESS_MODE}-png-512\" \n\ndef seed_everything(seed=42):\n    random.seed(seed)\n    np.random.seed(seed)\n\nseed_everything(SEED)\n\n\n# Hàm upload thư mục chứa file ZIP bằng rclone\ndef upload_folder_with_rclone(folder_path, remote_path, max_retries=3, retry_delay=10):\n    print(f\"📦 Đang đẩy {folder_path} lên {remote_path}...\")\n    \n    cmd = [\n        \"rclone\", \"copy\", \n        str(folder_path), \n        remote_path, \n        \"--transfers\", \"4\",\n        \"--checkers\", \"4\",\n        \"--retries\", \"3\",\n        \"--stats\", \"10s\"\n    ]\n    \n    for attempt in range(1, max_retries + 1):\n        try:\n            subprocess.run(cmd, check=True)\n            print(f\"✅ Đã upload thành công lên {remote_path} (Lần thử {attempt})\")\n            return\n        except subprocess.CalledProcessError as e:\n            print(f\"⚠️ [Lần thử {attempt}/{max_retries}] Lỗi rclone: {e}\")\n            if attempt < max_retries:\n                time.sleep(retry_delay)\n                retry_delay *= 2\n            else:\n                raise Exception(\"DỪNG CHƯƠNG TRÌNH: Upload thất bại sau nhiều lần thử.\")\n\n\ndef list_dicom_files(folder: Path):\n    files = []\n    for p in sorted(folder.rglob(\"*\")):\n        if p.is_file() and p.suffix.lower() in {\".dicom\", \".dcm\"}:\n            files.append(p)\n    return files\n\n\ndef _invert_if_monochrome1(ds, img):\n    if getattr(ds, \"PhotometricInterpretation\", \"\") == \"MONOCHROME1\":\n        img = np.max(img) - img\n    return img\n\n\ndef _normalize_to_uint8(img: np.ndarray):\n    img_min = float(img.min())\n    img_max = float(img.max())\n    if img_max > img_min:\n        img = (img - img_min) / (img_max - img_min)\n    else:\n        img = np.zeros_like(img, dtype=np.float32)\n    return (img * 255.0).clip(0, 255).astype(np.uint8)\n\n\n\ndef read_dicom_rescale_minmax(path: Path):\n    ds = pydicom.dcmread(str(path))\n    img = apply_modality_lut(ds.pixel_array, ds).astype(np.float32)\n    img = _invert_if_monochrome1(ds, img)\n    return _normalize_to_uint8(img)\n\ndef read_dicom_voilut(path: Path):\n    ds = pydicom.dcmread(str(path))\n    img = apply_voi_lut(ds.pixel_array, ds).astype(np.float32)\n    img = _invert_if_monochrome1(ds, img)\n    return _normalize_to_uint8(img)\n\n\n\ndef read_dicom_raw_minmax(path: Path):\n    ds = pydicom.dcmread(str(path))\n    img = ds.pixel_array.astype(np.float32)\n    img = _invert_if_monochrome1(ds, img)\n    return _normalize_to_uint8(img)\n\n\nPREPROCESS_FUNCS = {\n    \"raw_minmax\": read_dicom_raw_minmax,\n    \"rescale_minmax\": read_dicom_rescale_minmax,\n    \"voi_lut\": read_dicom_voilut\n}\n\n\nif PREPROCESS_MODE not in PREPROCESS_FUNCS:\n    raise ValueError(f\"PREPROCESS_MODE không hợp lệ: {PREPROCESS_MODE}\")\n\nread_dicom_to_uint8 = PREPROCESS_FUNCS[PREPROCESS_MODE]\nprint(\"Using PREPROCESS_MODE:\", PREPROCESS_MODE)\n\n\ndef resize_image_keep_shape(img, size=1024):\n    h, w = img.shape[:2]\n    resized = cv2.resize(img, (size, size), interpolation=cv2.INTER_AREA)\n    return resized, w, h\n\n\ndef sanitize_pascal_voc_boxes(boxes, width, height):\n    sanitized = []\n    for (x1, y1, x2, y2, class_id) in boxes:\n        x1 = float(np.clip(x1, 0, width))\n        y1 = float(np.clip(y1, 0, height))\n        x2 = float(np.clip(x2, 0, width))\n        y2 = float(np.clip(y2, 0, height))\n        if x2 <= x1 or y2 <= y1:\n            continue\n        sanitized.append((x1, y1, x2, y2, class_id))\n    return sanitized\n\n\ndef scale_boxes_to_imgsize(boxes, orig_w, orig_h, img_size):\n    scaled = []\n    if len(boxes) == 0:\n        return scaled\n    sx = img_size / orig_w\n    sy = img_size / orig_h\n    for (x1, y1, x2, y2, class_id) in boxes:\n        x1 = float(np.clip(x1 * sx, 0, img_size))\n        y1 = float(np.clip(y1 * sy, 0, img_size))\n        x2 = float(np.clip(x2 * sx, 0, img_size))\n        y2 = float(np.clip(y2 * sy, 0, img_size))\n        if x2 <= x1 or y2 <= y1:\n            continue\n        scaled.append((x1, y1, x2, y2, class_id))\n    return scaled\n\n\ndef save_image(img, out_path: Path):\n    if STACK_CHANNELS_3 and img.ndim == 2:\n        img = cv2.cvtColor(img, cv2.COLOR_GRAY2BGR)\n    if SAVE_AS_JPG:\n        cv2.imwrite(str(out_path), img, [cv2.IMWRITE_JPEG_QUALITY, JPG_QUALITY])\n    else:\n        cv2.imwrite(str(out_path), img, [cv2.IMWRITE_PNG_COMPRESSION, 0])\n\n\ndef process_one_dicom(dicom_path: Path, split: str, boxes):\n    image_id = dicom_path.stem\n    try:\n        img_raw = read_dicom_to_uint8(dicom_path)\n    except Exception as e:\n        print(f\"Error reading {dicom_path}: {e}\")\n        return []\n\n    samples = []\n    img_resized, orig_w, orig_h = resize_image_keep_shape(img_raw, size=IMG_SIZE)\n    boxes = sanitize_pascal_voc_boxes(boxes, width=orig_w, height=orig_h)\n    ext = \".jpg\" if SAVE_AS_JPG else \".png\"\n    out_name = f\"{image_id}{ext}\"\n    out_dir = IMG_TRAIN_DIR if split == \"train\" else IMG_VAL_DIR\n    out_path = out_dir / out_name\n    save_image(img_resized, out_path)\n    scaled_boxes = scale_boxes_to_imgsize(boxes, orig_w, orig_h, IMG_SIZE)\n    samples.append({\n        \"split\": split,\n        \"file_name\": f\"images/{split}/{out_name}\",\n        \"width\": IMG_SIZE,\n        \"height\": IMG_SIZE,\n        \"boxes\": scaled_boxes,\n    })\n    return samples\n\n\ndef build_coco(samples, class_names):\n    images = []\n    annotations = []\n    categories = [{\"id\": i + 1, \"name\": n} for i, n in enumerate(class_names)]\n    ann_id = 1\n    for img_id, s in enumerate(samples, start=1):\n        images.append({\n            \"id\": img_id,\n            \"file_name\": s[\"file_name\"],\n            \"width\": s[\"width\"],\n            \"height\": s[\"height\"],\n        })\n        for (x1, y1, x2, y2, class_id) in s[\"boxes\"]:\n            w, h = float(max(0.0, x2 - x1)), float(max(0.0, y2 - y1))\n            if w <= 0 or h <= 0:\n                continue\n            annotations.append({\n                \"id\": ann_id, \"image_id\": img_id, \"category_id\": int(class_id) + 1,\n                \"bbox\": [float(x1), float(y1), w, h], \"area\": w * h, \"iscrowd\": 0,\n            })\n            ann_id += 1\n    return {\"images\": images, \"annotations\": annotations, \"categories\": categories}\n\n\ndef chunker(seq, size):\n    return (seq[pos:pos + size] for pos in range(0, len(seq), size))\n\n\ndef main():\n    paths = [TRAIN_ANN_CSV, VAL_ANN_CSV, TRAIN_LABELS_CSV, VAL_LABELS_CSV]\n    for p in paths:\n        if not p.exists():\n            print(f\"File not found: {p}\")\n            return\n\n    df_train_ann = pd.read_csv(TRAIN_ANN_CSV)\n    df_val_ann = pd.read_csv(VAL_ANN_CSV)\n    df_train_labels = pd.read_csv(TRAIN_LABELS_CSV)\n    df_val_labels = pd.read_csv(VAL_LABELS_CSV)\n\n    class_names = sorted(list(set(df_train_ann[\"class_name\"].unique()) | set(df_val_ann[\"class_name\"].unique())))\n    if \"No finding\" in class_names:\n        class_names.remove(\"No finding\")\n    class2id = {name: i for i, name in enumerate(class_names)}\n\n    dicom_files_all = list_dicom_files(DATA_ROOT)\n    all_file_ids = {p.stem for p in dicom_files_all}\n    print(f\"Total DICOMs found in folder: {len(all_file_ids)}\")\n\n    train_ids_all = set(df_train_labels[\"image_id\"].unique()) & all_file_ids\n    val_ids_all = set(df_val_labels[\"image_id\"].unique()) & all_file_ids\n\n    remaining_ids = all_file_ids - train_ids_all - val_ids_all\n    if remaining_ids:\n        train_ids_all.update(remaining_ids)\n\n    id_to_split = {}\n    for x in train_ids_all:\n        id_to_split[x] = \"train\"\n    for x in val_ids_all:\n        id_to_split[x] = \"val\"\n\n    def build_ann_map(df):\n        ann_map = {}\n        df_objs = df[df[\"class_name\"].fillna(\"\").str.lower() != \"no finding\"]\n        for r in df_objs.itertuples(index=False):\n            if r.image_id in all_file_ids:\n                cid = class2id.get(r.class_name)\n                if cid is not None:\n                    ann_map.setdefault(r.image_id, []).append((r.x_min, r.y_min, r.x_max, r.y_max, cid))\n        return ann_map\n\n    train_ann_map = build_ann_map(df_train_ann)\n    val_ann_map = build_ann_map(df_val_ann)\n\n    to_process = [p for p in dicom_files_all if p.stem in id_to_split]\n    total_files = len(to_process)\n    print(f\"Images to process: {total_files} (Train: {len(train_ids_all)}, Val: {len(val_ids_all)})\")\n\n    max_workers = min(12, os.cpu_count() or 4)\n    print(f\"Using max_workers = {max_workers}\")\n\n    train_samples, val_samples = [], []\n    done_all = 0\n    start_all = time.perf_counter()\n\n    for batch_idx, batch_files in enumerate(chunker(to_process, BATCH_SIZE)):\n        print(f\"\\n--- BẮT ĐẦU XỬ LÝ MẺ {batch_idx + 1} ({len(batch_files)} ảnh) ---\")\n\n        with ThreadPoolExecutor(max_workers=max_workers) as executor:\n            futures = []\n            for p in batch_files:\n                image_id = p.stem\n                split = id_to_split[image_id]\n                boxes = train_ann_map.get(image_id, []) if split == \"train\" else val_ann_map.get(image_id, [])\n                futures.append(executor.submit(process_one_dicom, p, split, boxes))\n\n            for future in as_completed(futures):\n                try:\n                    samples = future.result()\n                    for s in samples:\n                        if s[\"split\"] == \"train\":\n                            train_samples.append(s)\n                        else:\n                            val_samples.append(s)\n                except Exception as e:\n                    print(\"Failed:\", e)\n\n                done_all += 1\n                if done_all % 500 == 0:\n                    print(f\"Tiến độ tổng: {done_all}/{total_files}...\")\n\n        zip_base_path = str(ZIP_DIR / f\"images_batch_{batch_idx + 1}\")\n        print(f\"🗜️ Đang nén mẻ ảnh thành file ZIP...\")\n        shutil.make_archive(zip_base_path, 'zip', str(OUT_ROOT))\n\n        upload_folder_with_rclone(folder_path=ZIP_DIR, remote_path=RCLONE_DEST)\n\n        print(\"🧹 Đang dọn dẹp file ZIP và ảnh local để giải phóng RAM/Disk...\")\n        for f in ZIP_DIR.glob(\"*.zip\"):\n            f.unlink()\n\n        for d in [IMG_TRAIN_DIR, IMG_VAL_DIR]:\n            for file_path in d.glob(\"*\"):\n                if file_path.is_file():\n                    file_path.unlink()\n\n        elapsed = time.perf_counter() - start_all\n        print(f\"✅ Hoàn thành mẻ {batch_idx + 1}. Tốc độ trung bình: {done_all/elapsed:.2f} img/s\")\n\n    print(\"\\n📝 Đã xử lý xong toàn bộ ảnh! Bắt đầu tạo file Annotations (JSON)...\")\n    with open(ANN_DIR / \"instances_train.json\", \"w\") as f:\n        json.dump(build_coco(train_samples, class_names), f)\n    with open(ANN_DIR / \"instances_val.json\", \"w\") as f:\n        json.dump(build_coco(val_samples, class_names), f)\n\n    print(\"🚀 Tải lên folder Annotations lên Google Drive (Lần cuối)...\")\n    upload_folder_with_rclone(folder_path=ANN_DIR, remote_path=f\"{RCLONE_DEST}/annotations\")\n\n    print(\"🎉 HOÀN THÀNH TOÀN BỘ QUY TRÌNH!\")\n\n\nif __name__ == \"__main__\":\n    main()\n","metadata":{"execution":{"iopub.status.busy":"2026-04-08T03:56:43.613946Z","iopub.execute_input":"2026-04-08T03:56:43.614636Z","iopub.status.idle":"2026-04-08T03:57:39.158692Z","shell.execute_reply.started":"2026-04-08T03:56:43.614601Z","shell.execute_reply":"2026-04-08T03:57:39.139435Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}