{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":1806494,"datasetId":1073296,"databundleVersionId":1843961},{"sourceType":"datasetVersion","sourceId":1810938,"datasetId":1075803,"databundleVersionId":1848422},{"sourceType":"datasetVersion","sourceId":16239523,"datasetId":10412878,"databundleVersionId":17222074},{"sourceType":"kernelVersion","sourceId":318851991}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"8aac49ff","cell_type":"markdown","source":"# VinBigData Inference on Train Set\n\nThis notebook loads the trained YOLO model, preprocesses the **training** images using the same pipeline as training, runs inference, and writes the `train_predictions.csv` mapping coordinates back to the original DICOM size. This is useful for evaluating the training accuracy or ensembling / pseudo-labeling.","metadata":{}},{"id":"488493b1","cell_type":"code","source":"!pip install ultralytics ensemble_boxes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T07:58:22.931651Z","iopub.execute_input":"2026-05-13T07:58:22.932364Z","iopub.status.idle":"2026-05-13T07:58:26.487151Z","shell.execute_reply.started":"2026-05-13T07:58:22.93233Z","shell.execute_reply":"2026-05-13T07:58:26.486455Z"}},"outputs":[],"execution_count":null},{"id":"aead62f4","cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom ultralytics import YOLO\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T07:58:26.489068Z","iopub.execute_input":"2026-05-13T07:58:26.489448Z","iopub.status.idle":"2026-05-13T07:58:26.493846Z","shell.execute_reply.started":"2026-05-13T07:58:26.489402Z","shell.execute_reply":"2026-05-13T07:58:26.493263Z"}},"outputs":[],"execution_count":null},{"id":"18deabea","cell_type":"markdown","source":"## 1. Configurations and Paths","metadata":{}},{"id":"740bfed7","cell_type":"code","source":"# Update these paths if running on Kaggle or a different local path\nPNG_DIR = \"/kaggle/input/datasets/xhlulu/vinbigdata-chest-xray-png-512px-original-ratio\"\nTRAIN_DIR = f\"{PNG_DIR}/train\"\n\n# Model weights path from training\nMODEL_PATH = \"/kaggle/working/runs/detect/vinbig-exp/yolo-dicom-wbf/weights/best.pt\"\n\n# Output predictions path\nOUTPUT_PATH = \"train_predictions.csv\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T07:58:26.494988Z","iopub.execute_input":"2026-05-13T07:58:26.495316Z","iopub.status.idle":"2026-05-13T07:58:26.507648Z","shell.execute_reply.started":"2026-05-13T07:58:26.495291Z","shell.execute_reply":"2026-05-13T07:58:26.506902Z"}},"outputs":[],"execution_count":null},{"id":"2dce3e7a","cell_type":"markdown","source":"## 2. Shared Preprocessing function\n\nWe must use exactly the same preprocessing function used during training to ensure the model sees identical data characteristics.","metadata":{}},{"id":"dbcdfdc7","cell_type":"code","source":"def preprocess_xray(img_bgr, clahe_clip=2.0, clahe_grid=(8, 8),\n                     denoise_h=5, sharpen_strength=0.6, target_size=512):\n    \"\"\"\n    Pipeline matching the training data preparation.\n    \"\"\"\n    gray = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2GRAY)\n    p_low, p_high = np.percentile(gray, (1, 99))\n    gray_norm = np.clip((gray.astype(np.float32) - p_low) /\n                        (p_high - p_low + 1e-6) * 255, 0, 255).astype(np.uint8)\n\n    clahe = cv2.createCLAHE(clipLimit=clahe_clip, tileGridSize=clahe_grid)\n    gray_clahe = clahe.apply(gray_norm)\n\n    if denoise_h > 0:\n        gray_denoised = cv2.fastNlMeansDenoising(\n            gray_clahe, None, h=denoise_h, templateWindowSize=7, searchWindowSize=21\n        )\n    else:\n        gray_denoised = gray_clahe\n\n    if sharpen_strength > 0:\n        blurred = cv2.GaussianBlur(gray_denoised, (0, 0), sigmaX=2.0)\n        sharpened = cv2.addWeighted(\n            gray_denoised, 1.0 + sharpen_strength,\n            blurred,        -sharpen_strength,\n            0\n        )\n        sharpened = np.clip(sharpened, 0, 255).astype(np.uint8)\n    else:\n        sharpened = gray_denoised\n\n    h_orig, w_orig = sharpened.shape[:2]\n    scale = min(target_size / w_orig, target_size / h_orig)\n    new_w = int(round(w_orig * scale))\n    new_h = int(round(h_orig * scale))\n    resized = cv2.resize(sharpened, (new_w, new_h), interpolation=cv2.INTER_LINEAR)\n\n    canvas = np.zeros((target_size, target_size), dtype=np.uint8)\n    pad_x = (target_size - new_w) // 2\n    pad_y = (target_size - new_h) // 2\n    canvas[pad_y:pad_y + new_h, pad_x:pad_x + new_w] = resized\n\n    processed_bgr = cv2.cvtColor(canvas, cv2.COLOR_GRAY2BGR)\n    return processed_bgr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T07:58:26.508709Z","iopub.execute_input":"2026-05-13T07:58:26.509049Z","iopub.status.idle":"2026-05-13T07:58:26.522156Z","shell.execute_reply.started":"2026-05-13T07:58:26.508993Z","shell.execute_reply":"2026-05-13T07:58:26.521536Z"}},"outputs":[],"execution_count":null},{"id":"d4b56eee","cell_type":"markdown","source":"## 3. Map Coordinates to Original\n\nPrediction bounding boxes must be transformed back to the original DICOM coordinate space.","metadata":{}},{"id":"0d9f7a39-7d5f-4bf7-a608-03108a49468a","cell_type":"code","source":"import os\n\n# tạo symbolic link từ input -> working\nos.symlink(\n    \"/kaggle/input/datasets/chinhde/yolov8/yolo_dataset\",\n    \"/kaggle/working/yolo_dataset\"\n)\n\nprint(\"Done\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T08:04:46.698599Z","iopub.execute_input":"2026-05-13T08:04:46.699265Z","iopub.status.idle":"2026-05-13T08:04:46.704044Z","shell.execute_reply.started":"2026-05-13T08:04:46.699232Z","shell.execute_reply":"2026-05-13T08:04:46.70327Z"}},"outputs":[],"execution_count":null},{"id":"48808e22","cell_type":"code","source":"# ==============================================================================\n# PHẦN ĐÁNH GIÁ METRICS (Precision, Recall, F1, mAP50, mAP75, mAP50-95)\n# ==============================================================================\n# Thay vì tự code inference chậm, YOLO có sẵn hàm validation CỰC NHANH.\n# Hàm này dùng PyTorch DataLoader (multi-process), tự đo mAP và F1 theo chuẩn COCO.\n\nDATASET_YAML = \"/kaggle/input/datasets/chinhde/yolov8/yolo_dataset/dataset.yaml\"\n\nprint(\"Đang chạy Evaluation trên tập Train...\")\nmetrics = model.val(\n    data=DATASET_YAML,\n    split='train',       # Chạy trên tập train\n    device=[0, 1],       # Dùng 2 GPU\n    batch=32,            # Tăng batch size để khai thác GPU\n    workers=8,           # Dùng 8 luồng CPU load data\n    conf=0.05,\n    iou=0.6,\n    verbose=False\n)\n\n# Extract metrics\np = metrics.results_dict['metrics/precision(B)']\nr = metrics.results_dict['metrics/recall(B)']\nf1 = 2 * (p * r) / (p + r + 1e-16)\nmap50 = metrics.results_dict['metrics/mAP50(B)']\nmap5095 = metrics.results_dict['metrics/mAP50-95(B)']\n# YOLO metrics/mAP object contains maps for IOU thresholds 0.50 to 0.95 with step 0.05\n# index 5 corresponds to 0.75 (0.50, 0.55, 0.60, 0.65, 0.70, 0.75)\nmap75 = metrics.box.maps[5]\n\nprint(\"=== KẾT QUẢ ĐÁNH GIÁ (TRAIN SET) ===\")\nprint(f\"Precision : {p:.4f}\")\nprint(f\"Recall    : {r:.4f}\")\nprint(f\"F1-Score  : {f1:.4f}\")\nprint(f\"mAP@50    : {map50:.4f}\")\nprint(f\"mAP@75    : {map75:.4f}\")\nprint(f\"mAP@50-95 : {map5095:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T08:04:49.024238Z","iopub.execute_input":"2026-05-13T08:04:49.024963Z","iopub.status.idle":"2026-05-13T08:06:38.659332Z","shell.execute_reply.started":"2026-05-13T08:04:49.024932Z","shell.execute_reply":"2026-05-13T08:06:38.658482Z"}},"outputs":[],"execution_count":null},{"id":"d7599579","cell_type":"code","source":"def convert_to_original(box, dim0, dim1, target_size=512):\n    \"\"\"\n    Inverse transform padded/scaled coordinates back to original DICOM dimensions.\n    \"\"\"\n    scale = min(target_size / dim1, target_size / dim0)\n    pad_x = (target_size - dim1 * scale) / 2\n    pad_y = (target_size - dim0 * scale) / 2\n    \n    x1, y1, x2, y2 = box\n    \n    # Remove padding and re-scale\n    x1 = (x1 - pad_x) / scale\n    y1 = (y1 - pad_y) / scale\n    x2 = (x2 - pad_x) / scale\n    y2 = (y2 - pad_y) / scale\n    \n    # Clip to true boundaries\n    x1 = max(0, min(x1, dim1))\n    y1 = max(0, min(y1, dim0))\n    x2 = max(0, min(x2, dim1))\n    y2 = max(0, min(y2, dim0))\n    \n    return [x1, y1, x2, y2]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T08:08:09.885935Z","iopub.execute_input":"2026-05-13T08:08:09.886168Z","iopub.status.idle":"2026-05-13T08:08:09.891467Z","shell.execute_reply.started":"2026-05-13T08:08:09.886144Z","shell.execute_reply":"2026-05-13T08:08:09.890772Z"}},"outputs":[],"execution_count":null},{"id":"0106bc78","cell_type":"markdown","source":"## 4. Map Coordinates & Save CSV (Cực nhanh qua Preprocessed Image)\nSử dụng trực tiếp ảnh `yolo_dataset/images/train` ĐÃ PREPROCESS ở notebook train (đỡ phần cv2 xử lý lại mất thời gian của CPU).","metadata":{}},{"id":"2b285e03","cell_type":"code","source":"import concurrent.futures\nimport math\n\n# Do đã preprocess ảnh và lưu dưới dạng 512x512 trong yolo_dataset\n# Chúng ta có thể pass thẳng thư mục này vào YOLO để GPU predict trực tiếp với DataLoader siêu nhanh!\nPREPROCESSED_TRAIN_DIR = \"/kaggle/working/yolo_dataset/images/train\"\n\ndf_meta = pd.read_csv(f\"{PNG_DIR}/train_meta.csv\")\nmeta_dict = df_meta.set_index('image_id')[['dim0', 'dim1']].to_dict('index')\n\nprint(\"Generating predictions via YOLO native batching...\")\n# YOLO predict trả về generator, xử lý song song với DataLoader\ngenerator = model.predict(source=PREPROCESSED_TRAIN_DIR, device=[0, 1], batch=32, conf=0.05, stream=True, verbose=False)\n\nresults = []\nfor preds in tqdm(generator, desc=\"Inference\"):\n    # Tên file ảnh (image_id.png) -> img_id\n    img_id = os.path.basename(preds.path).replace('.png', '')\n    \n    boxes = preds.boxes.xyxy.cpu().numpy()\n    scores = preds.boxes.conf.cpu().numpy()\n    classes = preds.boxes.cls.cpu().numpy()\n    \n    if len(boxes) == 0:\n        results.append({'image_id': img_id, 'PredictionString': '14 1 0 0 1 1'})\n    else:\n        # Lấy kích thước gốc từ Meta\n        dim0 = meta_dict.get(img_id, {}).get('dim0', 512)\n        dim1 = meta_dict.get(img_id, {}).get('dim1', 512)\n        \n        pred_strings = []\n        for box, score, cls in zip(boxes, scores, classes):\n            orig_box = convert_to_original(box, dim0, dim1)\n            pred_str = f\"{int(cls)} {score:.4f} {int(orig_box[0])} {int(orig_box[1])} {int(orig_box[2])} {int(orig_box[3])}\"\n            pred_strings.append(pred_str)\n        results.append({'image_id': img_id, 'PredictionString': \" \".join(pred_strings)})\n\n# Save to CSV\ndf_train_preds = pd.DataFrame(results)\ndf_train_preds.to_csv(OUTPUT_PATH, index=False)\nprint(\"Finished! Saved train predictions to:\", OUTPUT_PATH)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T08:08:09.892278Z","iopub.execute_input":"2026-05-13T08:08:09.892659Z","iopub.status.idle":"2026-05-13T08:09:42.37226Z","shell.execute_reply.started":"2026-05-13T08:08:09.892631Z","shell.execute_reply":"2026-05-13T08:09:42.371508Z"}},"outputs":[],"execution_count":null}]}