{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":14316647,"datasetId":9139361,"databundleVersionId":15122024},{"sourceType":"datasetVersion","sourceId":14312414,"datasetId":9136777,"databundleVersionId":15117402}],"dockerImageVersionId":31234,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"import os\nimport pandas as pd \nimport pydicom\nimport cv2\nimport numpy as np\nfrom tqdm import tqdm\nimport concurrent.futures\n\n# --- CONFIGURATION ---\nINPUT_DIR = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train\"\nOUTPUT_DIR = \"/kaggle/working/train_png_640\" # Recommend keeping naming clear\nFILTER_CSV = \"/kaggle/input/test-train-shorter-vindr/temp_train_1.csv\"\nTARGET_SIZE = 640  # 2160 is too heavy for training, 1024 is high-res enough\n\nos.makedirs(OUTPUT_DIR, exist_ok=True)\n\n# --- PROCESSING FUNCTION ---\ndef convert_one_image(filename):\n    if not filename.endswith('.dicom'): return\n    \n    file_path = os.path.join(INPUT_DIR, filename)\n    save_path = os.path.join(OUTPUT_DIR, filename.replace('.dicom', '.png'))\n    \n    # Resume check: Skip if file already exists\n    if os.path.exists(save_path): return\n\n    try:\n        # 1. Read DICOM file\n        dicom = pydicom.dcmread(file_path)\n        img = dicom.pixel_array\n        \n        # 2. Fix Monochrome1 (Inverted colors handling)\n        if \"PhotometricInterpretation\" in dicom and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n            img = np.amax(img) - img\n            \n        # 3. Apply Lung Windowing (Standard for Chest X-Ray)\n        # Center = -600, Width = 1500\n        # center, width = -600, 1500\n        # lower = center - width // 2\n        # upper = center + width // 2\n        # img = np.clip(img, lower, upper)\n        \n        # Normalize to 0-255\n        # img = (img - lower) / (upper - lower)\n        # img = (img * 255).astype(np.uint8)\n        \n        # 4. Resize with Padding (Letterbox) to preserve Aspect Ratio\n        h, w = img.shape[:2]\n        scale = min(TARGET_SIZE/h, TARGET_SIZE/w)\n        nh, nw = int(h*scale), int(w*scale)\n        \n        img_resized = cv2.resize(img, (nw, nh))\n        \n        # Create black canvas\n        final_img = np.zeros((TARGET_SIZE, TARGET_SIZE), dtype=np.uint8)\n        \n        # Paste resized image to center\n        dy = (TARGET_SIZE - nh) // 2\n        dx = (TARGET_SIZE - nw) // 2\n        final_img[dy:dy+nh, dx:dx+nw] = img_resized\n        \n        # Save to PNG\n        cv2.imwrite(save_path, final_img)\n        \n    except Exception as e:\n        print(f\"Error converting {filename}: {e}\")\n\n# --- MAIN EXECUTION ---\n\n# 1. Read required Image IDs from CSV\nprint(f\"Reading ID list from {FILTER_CSV}...\")\ndf_filter = pd.read_csv(FILTER_CSV)\nvalid_ids = set(df_filter['image_id'].unique()) # Use set for O(1) lookup\n\nprint(f\"Total images to process: {len(valid_ids)}\")\n\n# 2. Generate file list based on valid_ids\n# Note: CSV contains IDs (e.g., 'abc'), files are 'abc.dicom'\ntarget_files = [f\"{img_id}.dicom\" for img_id in valid_ids]\n\n# 3. Verify file existence (Safety check)\nprint(\"Verifying files in input directory...\")\nfiles_to_process = []\nfor f in target_files:\n    if os.path.exists(os.path.join(INPUT_DIR, f)):\n        files_to_process.append(f)\n    else:\n        # Rare case: ID in CSV but DICOM file missing\n        pass \n\nprint(f\"Starting multi-threaded processing for {len(files_to_process)} images...\")\n\n# 4. Run Multi-threading\n# Kaggle/Colab usually has 2-4 cores. \nwith concurrent.futures.ThreadPoolExecutor(max_workers=4) as executor:\n    list(tqdm(executor.map(convert_dicom_robust, files_to_process), total=len(files_to_process)))","metadata":{"_kg_hide-input":false}},{"cell_type":"markdown","source":"import pydicom\nimport cv2\nimport numpy as np\n\ndef convert_dicom_robust(dicom_path, target_size=640, fix_monochrome=True):\n    # 1. Đọc file\n    ds = pydicom.dcmread(dicom_path)\n    img = ds.pixel_array.astype(np.float32)\n\n    # 2. Xử lý Photometric Interpretation (Đảo màu nếu cần)\n    if fix_monochrome and hasattr(ds, \"PhotometricInterpretation\"):\n        if ds.PhotometricInterpretation == \"MONOCHROME1\":\n            # MONOCHROME1: 0 là Trắng -> Đảo lại thành 0 là Đen\n            img = np.amax(img) - img\n    \n    # 3. LOẠI BỎ WINDOWING CỨNG NHẮC (-600, 1500)\n    # Thay vào đó dùng ROBUST SCALING (Tự động tìm khoảng đẹp nhất của ảnh)\n    \n    # Lấy ngưỡng dưới (1%) và ngưỡng trên (99%)\n    # Bất kỳ điểm ảnh nào nằm ngoài khoảng này sẽ bị coi là nhiễu và cắt bỏ\n    p_low = np.percentile(img, 1)\n    p_high = np.percentile(img, 99)\n    \n    # Cắt giá trị (Clip)\n    img = np.clip(img, p_low, p_high)\n    \n    # 4. CHUẨN HÓA VỀ 0-255 (Dựa trên ngưỡng vừa tìm được)\n    if p_high > p_low:\n        img = (img - p_low) / (p_high - p_low)\n    else:\n        # Trường hợp hiếm: ảnh hỏng, toàn bộ 1 màu\n        img = np.zeros_like(img)\n\n    img = (img * 255.0).astype(np.uint8)\n\n    # 5. Resize + Padding (Giữ nguyên logic cũ của bạn)\n    h, w = img.shape\n    scale = min(target_size/h, target_size/w)\n    nh, nw = int(h*scale), int(w*scale)\n    img_resized = cv2.resize(img, (nw, nh))\n    \n    final_img = np.zeros((target_size, target_size), dtype=np.uint8)\n    dy = (target_size - nh) // 2\n    dx = (target_size - nw) // 2\n    final_img[dy:dy+nh, dx:dx+nw] = img_resized\n    \n    return final_img\n\n# --- Test thử ---\n# Thay đường dẫn đến 1 file dicom của bạn để test\ndicom_file = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/000434271f63a053c4128a0ba6352c7f.dicom\"\npng_img = convert_dicom_robust(dicom_file)\ncv2.imwrite(\"test_fixed_3.png\", png_img)\nprint(\"Đã lưu ảnh test_fixed.png. Hãy mở ra kiểm tra!\")","metadata":{"execution":{"iopub.status.busy":"2025-12-28T04:53:08.265111Z","iopub.execute_input":"2025-12-28T04:53:08.265585Z","iopub.status.idle":"2025-12-28T04:53:08.927644Z","shell.execute_reply.started":"2025-12-28T04:53:08.265551Z","shell.execute_reply":"2025-12-28T04:53:08.926364Z"}}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport pydicom\nimport cv2\nimport numpy as np\nfrom tqdm import tqdm\nimport concurrent.futures\n\n# --- CẤU HÌNH (SỬA LẠI NẾU CẦN) ---\n# Đường dẫn folder chứa ảnh DICOM gốc\nINPUT_DIR = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train\"\n\n# Folder đầu ra (Sẽ chứa toàn bộ file PNG)\nOUTPUT_DIR = \"/kaggle/working/train_png_640\" \n\n# File CSV lọc danh sách ảnh (như trong ảnh bạn gửi)\nFILTER_CSV = \"/kaggle/input/vindr-train-dims/train_v2(with_dims).csv\"  # Sửa đường dẫn này nếu file ở chỗ khác\n# Nếu không có file CSV lọc, có thể comment dòng trên và dùng os.listdir\n\nTARGET_SIZE = 1024 # Size ảnh mong muốn\nos.makedirs(OUTPUT_DIR, exist_ok=True)\n\n# --- 1. HÀM XỬ LÝ ẢNH \"ROBUST\" (BẤT CHẤP NHIỄU) ---\ndef convert_dicom_robust(dicom_path, target_size=640, fix_monochrome=True):\n    try:\n        ds = pydicom.dcmread(dicom_path)\n        img = ds.pixel_array.astype(np.float32)\n\n        # Fix lỗi âm bản\n        if fix_monochrome and hasattr(ds, \"PhotometricInterpretation\"):\n            if ds.PhotometricInterpretation == \"MONOCHROME1\":\n                img = np.amax(img) - img\n        \n        # Robust Scaling (Thay cho Windowing cứng nhắc)\n        # Cắt bỏ 1% điểm ảnh quá sáng/quá tối để loại nhiễu\n        p_low = np.percentile(img, 1)\n        p_high = np.percentile(img, 99)\n        img = np.clip(img, p_low, p_high)\n        \n        # Chuẩn hóa Min-Max về 0-1\n        if p_high > p_low:\n            img = (img - p_low) / (p_high - p_low)\n        else:\n            img = np.zeros_like(img) # Ảnh lỗi\n\n        # Đưa về 0-255\n        img = (img * 255.0).astype(np.uint8)\n\n        # Resize + Padding (Letterbox)\n        h, w = img.shape\n        scale = min(target_size/h, target_size/w)\n        nh, nw = int(h*scale), int(w*scale)\n        img_resized = cv2.resize(img, (nw, nh))\n        \n        final_img = np.zeros((target_size, target_size), dtype=np.uint8)\n        dy = (target_size - nh) // 2\n        dx = (target_size - nw) // 2\n        final_img[dy:dy+nh, dx:dx+nw] = img_resized\n        \n        return final_img\n    except Exception as e:\n        # Nếu file dicom bị lỗi đọc, trả về None\n        return None\n\n# --- 2. HÀM WRAPPER: KẾT NỐI INPUT -> XỬ LÝ -> OUTPUT ---\ndef process_and_save(filename):\n    # Đảm bảo filename có đuôi .dicom\n    if not filename.endswith('.dicom'): \n        filename = f\"{filename}.dicom\"\n        \n    # TẠO ĐƯỜNG DẪN ĐẦY ĐỦ (Fix lỗi FileNot Found)\n    in_path = os.path.join(INPUT_DIR, filename)\n    out_path = os.path.join(OUTPUT_DIR, filename.replace('.dicom', '.png'))\n    \n    # Skip nếu file đã tồn tại (để resume nếu bị ngắt)\n    if os.path.exists(out_path): return\n\n    # Gọi hàm xử lý\n    processed_img = convert_dicom_robust(in_path, target_size=TARGET_SIZE)\n    \n    # Lưu ảnh nếu xử lý thành công\n    if processed_img is not None:\n        cv2.imwrite(out_path, processed_img)\n\n# --- 3. CHƯƠNG TRÌNH CHÍNH ---\n\n# Lấy danh sách file cần làm\nif os.path.exists(FILTER_CSV):\n    print(f\"Đang đọc danh sách ID từ {FILTER_CSV}...\")\n    df = pd.read_csv(FILTER_CSV)\n    # Lấy cột image_id và đảm bảo thêm đuôi .dicom\n    target_files = [f\"{img_id}.dicom\" if not str(img_id).endswith('.dicom') else str(img_id) \n                    for img_id in df['image_id'].unique()]\nelse:\n    print(\"Không tìm thấy CSV lọc, sẽ quét toàn bộ thư mục input...\")\n    target_files = [f for f in os.listdir(INPUT_DIR) if f.endswith('.dicom')]\n\n# Lọc lại lần cuối để chắc chắn file input có tồn tại\nprint(\"Đang kiểm tra file input...\")\nfiles_to_process = [f for f in target_files if os.path.exists(os.path.join(INPUT_DIR, f))]\n\nprint(f\"--> BẮT ĐẦU XỬ LÝ {len(files_to_process)} ẢNH...\")\nprint(f\"--> OUTPUT SẼ LƯU TẠI: {OUTPUT_DIR}\")\n\n# Chạy đa luồng (4 workers là tối ưu cho Kaggle)\nwith concurrent.futures.ThreadPoolExecutor(max_workers=4) as executor:\n    list(tqdm(executor.map(process_and_save, files_to_process), total=len(files_to_process)))\n    \nprint(\"\\n=== HOÀN TẤT! ===\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-28T05:04:48.728956Z","iopub.execute_input":"2025-12-28T05:04:48.729366Z","iopub.status.idle":"2025-12-28T05:18:44.818876Z","shell.execute_reply.started":"2025-12-28T05:04:48.729334Z","shell.execute_reply":"2025-12-28T05:18:44.817294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nimport os\n\n# --- CẤU HÌNH ---\nSOURCE_DIR = \"/kaggle/working/train_png_640\" # Folder chứa ảnh PNG vừa tạo\nOUTPUT_NAME = \"/kaggle/working/train_png_640\" # Tên file zip đầu ra\n\nprint(\"Đang nén file zip... (Bước này mất vài phút)\")\n# Tạo file zip: vinbigdata_preprocessed.zip\nshutil.make_archive(OUTPUT_NAME, 'zip', SOURCE_DIR)\n\nprint(f\"Xong! File đã lưu tại: {OUTPUT_NAME}.zip\")\n\n# (Tùy chọn) Xóa folder ảnh lẻ đi cho nhẹ Output, chỉ giữ lại file zip\n# shutil.rmtree(SOURCE_DIR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-28T05:19:43.773494Z","iopub.execute_input":"2025-12-28T05:19:43.774192Z","iopub.status.idle":"2025-12-28T05:19:53.347008Z","shell.execute_reply.started":"2025-12-28T05:19:43.774141Z","shell.execute_reply":"2025-12-28T05:19:53.345901Z"}},"outputs":[],"execution_count":null}]}