{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":15094918,"datasetId":9664561,"databundleVersionId":15979687},{"sourceType":"datasetVersion","sourceId":15587920,"datasetId":9973498,"databundleVersionId":16520173}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!curl https://rclone.org/install.sh | sudo bash\n!mkdir -p /root/.config/rclone\n!cp /kaggle/input/datasets/bacahlam/sdsdsdsd/rclone.conf /root/.config/rclone/rclone.conf\n!rclone lsd rclone:","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-08T06:16:23.443242Z","iopub.execute_input":"2026-04-08T06:16:23.444044Z","iopub.status.idle":"2026-04-08T06:16:29.732531Z","shell.execute_reply.started":"2026-04-08T06:16:23.444009Z","shell.execute_reply":"2026-04-08T06:16:29.731813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport json\nimport time\nimport shutil\nimport subprocess\nimport random\nfrom pathlib import Path\nfrom concurrent.futures import ThreadPoolExecutor, as_completed\n\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport pydicom\n\n# =========================\n# PATH CONFIG\n# =========================\nDATA_ROOT = Path(\"/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection\")\nTRAIN_ANN_CSV = Path(\"/kaggle/input/datasets/benxelua/correct-label/annotations/annotations_train.csv\")\nVAL_ANN_CSV = Path(\"/kaggle/input/datasets/benxelua/correct-label/annotations/annotations_test.csv\")\nTRAIN_LABELS_CSV = Path(\"/kaggle/input/datasets/benxelua/correct-label/annotations/image_labels_train.csv\")\nVAL_LABELS_CSV = Path(\"/kaggle/input/datasets/benxelua/correct-label/annotations/image_labels_test.csv\")\n\n# Đặt thẳng tên thư mục là tên dataset để không cần rename ở cuối\nOUT_ROOT = Path(\"vinbigdata-adaptive-preprocessed-512-2\")\nIMG_DIR = OUT_ROOT / \"images\"\nANN_DIR = OUT_ROOT / \"annotations\"\n\nIMG_TRAIN_DIR = IMG_DIR / \"train\"\nIMG_VAL_DIR = IMG_DIR / \"val\"\n\n# Thư mục tạm để chứa file ZIP trước khi đẩy lên rclone\nZIP_DIR = Path(\"temp_zips\")\n\nfor d in [IMG_TRAIN_DIR, IMG_VAL_DIR, ANN_DIR, ZIP_DIR]:\n    d.mkdir(parents=True, exist_ok=True)\n\n# =========================\n# IMAGE CONFIG\n# =========================\nIMG_SIZE = 512\nSAVE_AS_JPG = False\nJPG_QUALITY = 95\nSTACK_CHANNELS_3 = True\nPNG_COMPRESSION = 0\n\n# =========================\n# ADAPTIVE NORMALIZATION CONFIG\n# =========================\nX_CDF_LOW = 0.05\nX_CDF_HIGH = 0.95\nY_CDF_LOW = 0.15\nY_CDF_HIGH = 0.95\nTARGET_MEAN = 0.4776 * 255.0   # ~121.8\nTARGET_STD = 0.2238 * 255.0    # ~57.1\nEPS = 1e-6\n\n# =========================\n# RCLONE BATCH UPLOAD CONFIG\n# =========================\nBATCH_SIZE = 3000\n# Thư mục đích trên Google Drive (Rclone sẽ tự tạo nếu chưa có)\nRCLONE_DEST = \"rclone:vinbigdata-adaptive-preprocessed-512-2\" \n\ndef seed_everything(seed=42):\n    random.seed(seed)\n    np.random.seed(seed)\n\nseed_everything(42)\n\n\n# =========================\n# HÀM UPLOAD BẰNG RCLONE (An toàn, có Retry)\n# =========================\ndef upload_folder_with_rclone(folder_path, remote_path, max_retries=3, retry_delay=10):\n    print(f\"📦 Đang đẩy {folder_path} lên {remote_path} bằng rclone...\")\n    \n    cmd = [\n        \"rclone\", \"copy\", \n        str(folder_path), \n        remote_path, \n        \"--transfers\", \"4\",         # Copy song song 4 file ZIP \n        \"--checkers\", \"4\",\n        \"--retries\", \"3\",           \n        \"--stats\", \"10s\"            \n    ]\n    \n    for attempt in range(1, max_retries + 1):\n        try:\n            subprocess.run(cmd, check=True)\n            print(f\"✅ Đã upload thành công lên {remote_path} (Lần thử {attempt})\")\n            return\n        except subprocess.CalledProcessError as e:\n            print(f\"⚠️ [Lần thử {attempt}/{max_retries}] Lỗi rclone: {e}\")\n            if attempt < max_retries:\n                time.sleep(retry_delay)\n                retry_delay *= 2 \n            else:\n                raise Exception(\"❌ DỪNG CHƯƠNG TRÌNH: Upload thất bại sau nhiều lần thử, bảo toàn dữ liệu local.\")\n\n\ndef list_dicom_files(folder: Path):\n    files = []\n    for p in sorted(folder.rglob(\"*\")):\n        if p.is_file() and p.suffix.lower() in {\".dicom\", \".dcm\"}:\n            files.append(p)\n    return files\n\n\ndef read_dicom_to_float(path: Path):\n    ds = pydicom.dcmread(str(path))\n    img = ds.pixel_array.astype(np.float32)\n\n    if getattr(ds, \"PhotometricInterpretation\", \"\") == \"MONOCHROME1\":\n        img = np.max(img) - img\n\n    slope = float(getattr(ds, \"RescaleSlope\", 1.0))\n    intercept = float(getattr(ds, \"RescaleIntercept\", 0.0))\n    img = img * slope + intercept\n    return img\n\n\ndef compute_cdf_crop_bounds(img: np.ndarray, x_low=0.05, x_high=0.95, y_low=0.15, y_high=0.95):\n    h, w = img.shape\n    x_profile = img.sum(axis=0).astype(np.float64)\n    y_profile = img.sum(axis=1).astype(np.float64)\n\n    if x_profile.sum() <= 0:\n        x1, x2 = 0, w\n    else:\n        x_cdf = np.cumsum(x_profile) / (x_profile.sum() + EPS)\n        x1 = int(np.searchsorted(x_cdf, x_low, side=\"left\"))\n        x2 = int(np.searchsorted(x_cdf, x_high, side=\"right\"))\n\n    if y_profile.sum() <= 0:\n        y1, y2 = 0, h\n    else:\n        y_cdf = np.cumsum(y_profile) / (y_profile.sum() + EPS)\n        y1 = int(np.searchsorted(y_cdf, y_low, side=\"left\"))\n        y2 = int(np.searchsorted(y_cdf, y_high, side=\"right\"))\n\n    x1 = max(0, min(x1, w - 1))\n    x2 = max(x1 + 1, min(x2, w))\n    y1 = max(0, min(y1, h - 1))\n    y2 = max(y1 + 1, min(y2, h))\n\n    return x1, y1, x2, y2\n\n\ndef adaptive_normalize_image(img: np.ndarray, target_mean=121.8, target_std=57.1):\n    mu_orig = float(img.mean())\n    sigma_orig = float(img.std())\n    if sigma_orig < EPS:\n        sigma_orig = 1.0\n\n    img_norm = ((img - mu_orig) / sigma_orig) * target_std + target_mean\n    img_norm = np.clip(img_norm, 0, 255).astype(np.uint8)\n    return img_norm\n\n\ndef read_dicom_adaptive_normalization(path: Path):\n    img = read_dicom_to_float(path)\n    x1, y1, x2, y2 = compute_cdf_crop_bounds(\n        img, x_low=X_CDF_LOW, x_high=X_CDF_HIGH, y_low=Y_CDF_LOW, y_high=Y_CDF_HIGH,\n    )\n    cropped = img[y1:y2, x1:x2]\n\n    if cropped.size == 0:\n        x1, y1, x2, y2 = 0, 0, img.shape[1], img.shape[0]\n        cropped = img\n\n    img_norm = adaptive_normalize_image(cropped, target_mean=TARGET_MEAN, target_std=TARGET_STD)\n\n    crop_info = {\n        \"crop_x1\": x1, \"crop_y1\": y1, \"crop_x2\": x2, \"crop_y2\": y2,\n        \"cropped_w\": x2 - x1, \"cropped_h\": y2 - y1,\n        \"orig_w\": img.shape[1], \"orig_h\": img.shape[0],\n    }\n    return img_norm, crop_info\n\n\ndef resize_image_keep_shape(img, size=1024):\n    h, w = img.shape[:2]\n    resized = cv2.resize(img, (size, size), interpolation=cv2.INTER_AREA)\n    return resized, w, h\n\n\ndef sanitize_pascal_voc_boxes(boxes, width, height):\n    sanitized = []\n    for (x1, y1, x2, y2, class_id) in boxes:\n        x1 = float(np.clip(x1, 0, width))\n        y1 = float(np.clip(y1, 0, height))\n        x2 = float(np.clip(x2, 0, width))\n        y2 = float(np.clip(y2, 0, height))\n        if x2 <= x1 or y2 <= y1:\n            continue\n        sanitized.append((x1, y1, x2, y2, class_id))\n    return sanitized\n\n\ndef crop_boxes_to_roi(boxes, crop_x1, crop_y1, crop_x2, crop_y2):\n    cropped_boxes = []\n    crop_w = crop_x2 - crop_x1\n    crop_h = crop_y2 - crop_y1\n\n    for (x1, y1, x2, y2, class_id) in boxes:\n        nx1, ny1 = x1 - crop_x1, y1 - crop_y1\n        nx2, ny2 = x2 - crop_x1, y2 - crop_y1\n\n        nx1 = float(np.clip(nx1, 0, crop_w))\n        ny1 = float(np.clip(ny1, 0, crop_h))\n        nx2 = float(np.clip(nx2, 0, crop_w))\n        ny2 = float(np.clip(ny2, 0, crop_h))\n\n        if nx2 <= nx1 or ny2 <= ny1:\n            continue\n        cropped_boxes.append((nx1, ny1, nx2, ny2, class_id))\n    return cropped_boxes\n\n\ndef scale_boxes_to_imgsize(boxes, orig_w, orig_h, img_size):\n    scaled = []\n    if len(boxes) == 0: return scaled\n    sx, sy = img_size / orig_w, img_size / orig_h\n    for (x1, y1, x2, y2, class_id) in boxes:\n        x1, y1 = float(np.clip(x1 * sx, 0, img_size)), float(np.clip(y1 * sy, 0, img_size))\n        x2, y2 = float(np.clip(x2 * sx, 0, img_size)), float(np.clip(y2 * sy, 0, img_size))\n        if x2 <= x1 or y2 <= y1: continue\n        scaled.append((x1, y1, x2, y2, class_id))\n    return scaled\n\n\ndef save_image(img, out_path: Path):\n    if STACK_CHANNELS_3 and img.ndim == 2:\n        img = cv2.cvtColor(img, cv2.COLOR_GRAY2BGR)\n    if SAVE_AS_JPG:\n        cv2.imwrite(str(out_path), img, [cv2.IMWRITE_JPEG_QUALITY, JPG_QUALITY])\n    else:\n        cv2.imwrite(str(out_path), img, [cv2.IMWRITE_PNG_COMPRESSION, PNG_COMPRESSION])\n\n\ndef process_one_dicom(dicom_path: Path, split: str, boxes):\n    image_id = dicom_path.stem\n    try:\n        # 1. Đọc ảnh, chạy CDF Crop và Adaptive Normalization\n        # img_adapt lúc này có kích thước là (crop_h, crop_w)\n        img_adapt, crop_info = read_dicom_adaptive_normalization(dicom_path)\n    except Exception as e:\n        print(f\"Error reading {dicom_path}: {e}\")\n        return []\n\n    # 2. Cắt tọa độ box theo vùng ROI đã crop (tọa độ lúc này thuộc về vùng crop)\n    boxes_cropped = crop_boxes_to_roi(\n        boxes,\n        crop_x1=crop_info[\"crop_x1\"], \n        crop_y1=crop_info[\"crop_y1\"],\n        crop_x2=crop_info[\"crop_x2\"], \n        crop_y2=crop_info[\"crop_y2\"],\n    )\n    \n    # 3. Loại bỏ box lỗi (nếu có)\n    boxes_cropped = sanitize_pascal_voc_boxes(\n        boxes_cropped, \n        width=crop_info[\"cropped_w\"], \n        height=crop_info[\"cropped_h\"]\n    )\n\n    # 4. Resize ảnh từ kích thước crop về 512\n    img_resized, _, _ = resize_image_keep_shape(img_adapt, size=IMG_SIZE)\n\n    # 5. FIX LỖI SCALE: \n    # orig_w và orig_h PHẢI là kích thước của vùng sau khi CROP (cropped_w, cropped_h)\n    # img_size là 512\n    scaled_boxes = scale_boxes_to_imgsize(\n        boxes_cropped, \n        orig_w=crop_info[\"cropped_w\"], \n        orig_h=crop_info[\"cropped_h\"], \n        img_size=IMG_SIZE\n    )\n\n    # Lưu ảnh\n    ext = \".jpg\" if SAVE_AS_JPG else \".png\"\n    out_name = f\"{image_id}{ext}\"\n    out_dir = IMG_TRAIN_DIR if split == \"train\" else IMG_VAL_DIR\n    out_path = out_dir / out_name\n    save_image(img_resized, out_path)\n\n    return [{\n        \"split\": split,\n        \"file_name\": f\"images/{split}/{out_name}\",\n        \"width\": IMG_SIZE,\n        \"height\": IMG_SIZE,\n        \"boxes\": scaled_boxes,\n    }]\n\ndef build_coco(samples, class_names):\n    images, annotations = [], []\n    categories = [{\"id\": i + 1, \"name\": n} for i, n in enumerate(class_names)]\n\n    ann_id = 1\n    for img_id, s in enumerate(samples, start=1):\n        images.append({\"id\": img_id, \"file_name\": s[\"file_name\"], \"width\": s[\"width\"], \"height\": s[\"height\"]})\n        for (x1, y1, x2, y2, class_id) in s[\"boxes\"]:\n            w, h = float(max(0.0, x2 - x1)), float(max(0.0, y2 - y1))\n            if w <= 0 or h <= 0: continue\n            annotations.append({\n                \"id\": ann_id, \"image_id\": img_id, \"category_id\": int(class_id) + 1,\n                \"bbox\": [float(x1), float(y1), w, h], \"area\": w * h, \"iscrowd\": 0,\n            })\n            ann_id += 1\n\n    return {\"images\": images, \"annotations\": annotations, \"categories\": categories}\n\n\n# Hàm chia batch\ndef chunker(seq, size):\n    return (seq[pos:pos + size] for pos in range(0, len(seq), size))\n\ndef main():\n    paths = [TRAIN_ANN_CSV, VAL_ANN_CSV, TRAIN_LABELS_CSV, VAL_LABELS_CSV]\n    for p in paths:\n        if not p.exists():\n            print(f\"File not found: {p}\")\n            return\n\n    df_train_ann = pd.read_csv(TRAIN_ANN_CSV)\n    df_val_ann = pd.read_csv(VAL_ANN_CSV)\n    df_train_labels = pd.read_csv(TRAIN_LABELS_CSV)\n    df_val_labels = pd.read_csv(VAL_LABELS_CSV)\n\n    class_names = sorted(list(set(df_train_ann[\"class_name\"].unique()) | set(df_val_ann[\"class_name\"].unique())))\n    if \"No finding\" in class_names: class_names.remove(\"No finding\")\n    class2id = {name: i for i, name in enumerate(class_names)}\n\n    dicom_files_all = list_dicom_files(DATA_ROOT)\n    all_file_ids = {p.stem for p in dicom_files_all}\n    print(f\"Total DICOMs found in folder: {len(all_file_ids)}\")\n\n    train_ids_all = set(df_train_labels[\"image_id\"].astype(str).unique()) & all_file_ids\n    val_ids_all = set(df_val_labels[\"image_id\"].astype(str).unique()) & all_file_ids\n\n    remaining_ids = all_file_ids - train_ids_all - val_ids_all\n    if remaining_ids:\n        train_ids_all.update(remaining_ids)\n\n    id_to_split = {}\n    for x in train_ids_all: id_to_split[x] = \"train\"\n    for x in val_ids_all: id_to_split[x] = \"val\"\n\n    def build_ann_map(df):\n        ann_map = {}\n        df_objs = df[df[\"class_name\"].fillna(\"\").str.lower() != \"no finding\"].copy()\n        for r in df_objs.itertuples(index=False):\n            if str(r.image_id) in all_file_ids:\n                cid = class2id.get(r.class_name)\n                if cid is not None:\n                    ann_map.setdefault(str(r.image_id), []).append((float(r.x_min), float(r.y_min), float(r.x_max), float(r.y_max), cid))\n        return ann_map\n\n    train_ann_map = build_ann_map(df_train_ann)\n    val_ann_map = build_ann_map(df_val_ann)\n\n    to_process = [p for p in dicom_files_all if p.stem in id_to_split]\n    total_files = len(to_process)\n    print(f\"Images to process: {total_files} (Train: {len(train_ids_all)}, Val: {len(val_ids_all)})\")\n\n    max_workers = min(8, os.cpu_count())\n    print(f\"Using max_workers = {max_workers}\")\n\n    train_samples, val_samples = [], []\n    done_all = 0\n    start_all = time.perf_counter()\n\n    # ====================================================================\n    # LẶP QUA TỪNG BATCH: XỬ LÝ -> ZIP -> UPLOAD -> XÓA DỌN DẸP\n    # ====================================================================\n    for batch_idx, batch_files in enumerate(chunker(to_process, BATCH_SIZE)):\n        print(f\"\\n--- BẮT ĐẦU XỬ LÝ MẺ {batch_idx + 1} ({len(batch_files)} ảnh) ---\")\n        \n        with ThreadPoolExecutor(max_workers=max_workers) as executor:\n            futures = []\n            for p in batch_files:\n                image_id = p.stem\n                split = id_to_split[image_id]\n                boxes = train_ann_map.get(image_id, []) if split == \"train\" else val_ann_map.get(image_id, [])\n                futures.append(executor.submit(process_one_dicom, p, split, boxes))\n\n            for future in as_completed(futures):\n                try:\n                    samples = future.result()\n                    for s in samples:\n                        if s[\"split\"] == \"train\": train_samples.append(s)\n                        else: val_samples.append(s)\n                except Exception as e:\n                    print(\"Failed:\", e)\n                \n                done_all += 1\n                if done_all % 500 == 0:\n                    print(f\"Tiến độ tổng: {done_all}/{total_files}...\")\n\n        # 1. Nén folder OUT_ROOT thành file ZIP\n        zip_base_path = str(ZIP_DIR / f\"images_batch_{batch_idx + 1}\")\n        print(f\"🗜️ Đang nén mẻ ảnh thành file ZIP...\")\n        shutil.make_archive(zip_base_path, 'zip', str(OUT_ROOT))\n\n        # 2. Upload file ZIP lên GG Drive qua Rclone\n        print(f\"\\n🚀 Đang tải file ZIP của mẻ thứ {batch_idx + 1} lên Google Drive...\")\n        upload_folder_with_rclone(folder_path=ZIP_DIR, remote_path=RCLONE_DEST)\n\n        # 3. Dọn dẹp sạch sẽ\n        print(\"🧹 Đang dọn dẹp file ZIP và ảnh local để giải phóng RAM/Disk...\")\n        # Xóa file ZIP vừa up\n        for f in ZIP_DIR.glob(\"*.zip\"):\n            f.unlink()\n            \n        # Xóa ảnh ở mẻ này\n        for d in [IMG_TRAIN_DIR, IMG_VAL_DIR]:\n            for file_path in d.glob(\"*\"):\n                if file_path.is_file():\n                    file_path.unlink() \n                    \n        elapsed = time.perf_counter() - start_all\n        print(f\"✅ Hoàn thành mẻ {batch_idx + 1}. Tốc độ trung bình: {done_all/elapsed:.2f} img/s\")\n\n\n    # ====================================================================\n    # BƯỚC CUỐI: TẠO FILE JSON COCO VÀ UPLOAD\n    # ====================================================================\n    print(\"\\n📝 Đã xử lý xong toàn bộ ảnh! Bắt đầu tạo file Annotations (JSON)...\")\n    with open(ANN_DIR / \"instances_train.json\", \"w\") as f:\n        json.dump(build_coco(train_samples, class_names), f)\n    with open(ANN_DIR / \"instances_val.json\", \"w\") as f:\n        json.dump(build_coco(val_samples, class_names), f)\n\n    print(\"🚀 Tải lên folder Annotations lên Google Drive (Lần cuối)...\")\n    # Upload thẳng folder annotations/ không cần nén\n    upload_folder_with_rclone(folder_path=ANN_DIR, remote_path=f\"{RCLONE_DEST}/annotations\")\n\n    print(f\"🎉 Hoàn thành toàn bộ quy trình đẩy lên Google Drive!\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}