{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":16226735,"datasetId":10404334,"databundleVersionId":17208131}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"    working/\n    ├── vinbig_yolo/\n    │\n    ├── images/\n    │   ├── train/\n    │   ├── val/\n    │\n    ├── labels/\n    │   ├── train/\n    │   ├── val/\n    │\n    ├── data.yaml","metadata":{}},{"cell_type":"code","source":"pip install dicomsdl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T14:38:35.308135Z","iopub.execute_input":"2026-05-12T14:38:35.308431Z","iopub.status.idle":"2026-05-12T14:38:41.771971Z","shell.execute_reply.started":"2026-05-12T14:38:35.308397Z","shell.execute_reply":"2026-05-12T14:38:41.771121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport dicomsdl\nimport numpy as np\nimport pandas as pd\n\nfrom tqdm import tqdm\nfrom concurrent.futures import ProcessPoolExecutor\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T14:38:41.774293Z","iopub.execute_input":"2026-05-12T14:38:41.774634Z","iopub.status.idle":"2026-05-12T14:38:43.824425Z","shell.execute_reply.started":"2026-05-12T14:38:41.774595Z","shell.execute_reply":"2026-05-12T14:38:43.823545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # =====================================================\n# # CONFIG\n# # =====================================================\n\n# ROOT = \"/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection\"\n\n# TRAIN_DIR = f\"{ROOT}/train\"\n\n# CSV_PATH = f\"{ROOT}/train.csv\"\n\n# OUT_ROOT = \"/kaggle/working/vinbig_yolo\"\n\n# IMG_SIZE = 768\n# JPEG_QUALITY = 85\n\n# NUM_WORKERS = 4\n\n# # =====================================================\n# # OUTPUT\n# # =====================================================\n\n# for split in [\"train\", \"val\"]:\n\n#     os.makedirs(\n#         f\"{OUT_ROOT}/images/{split}\",\n#         exist_ok=True\n#     )\n\n#     os.makedirs(\n#         f\"{OUT_ROOT}/labels/{split}\",\n#         exist_ok=True\n#     )\n\n# # =====================================================\n# # LOAD CSV\n# # =====================================================\n\n# df = pd.read_csv(CSV_PATH)\n\n# grouped = df.groupby(\"image_id\")\n\n# image_ids = list(grouped.groups.keys())\n\n# # =====================================================\n# # SPLIT\n# # =====================================================\n\n# train_ids, val_ids = train_test_split(\n#     image_ids,\n#     test_size=0.2,\n#     random_state=42\n# )\n\n# train_set = set(train_ids)\n\n# # =====================================================\n# # CLAHE\n# # =====================================================\n\n# clahe = cv2.createCLAHE(\n#     clipLimit=2.0,\n#     tileGridSize=(4,4)\n# )\n\n# # =====================================================\n# # NORMALIZE\n# # =====================================================\n\n# def normalize(img):\n\n#     img = img.astype(np.float32)\n\n#     img -= img.min()\n\n#     img /= (img.max() + 1e-6)\n\n#     img *= 255\n\n#     return img.astype(np.uint8)\n\n# # =====================================================\n# # PROCESS ONE\n# # =====================================================\n\n# def process_one(image_id):\n\n#     try:\n\n#         dicom_path = (\n#             f\"{TRAIN_DIR}/{image_id}.dicom\"\n#         )\n\n#         ds = dicomsdl.open(dicom_path)\n\n#         img = ds.pixelData()\n\n#         h0, w0 = img.shape\n\n#         img = normalize(img)\n\n#         img = clahe.apply(img)\n\n#         img = cv2.resize(\n#             img,\n#             (IMG_SIZE, IMG_SIZE),\n#             interpolation=cv2.INTER_AREA\n#         )\n\n#         split = (\n#             \"train\"\n#             if image_id in train_set\n#             else \"val\"\n#         )\n\n#         # =================================================\n#         # SAVE IMAGE\n#         # =================================================\n\n#         img_out = (\n#             f\"{OUT_ROOT}/images/\"\n#             f\"{split}/{image_id}.jpg\"\n#         )\n\n#         cv2.imwrite(\n#             img_out,\n#             img,\n#             [\n#                 cv2.IMWRITE_JPEG_QUALITY,\n#                 JPEG_QUALITY\n#             ]\n#         )\n\n#         # =================================================\n#         # LABEL\n#         # =================================================\n\n#         rows = grouped.get_group(image_id)\n\n#         lines = []\n\n#         for _, row in rows.iterrows():\n\n#             cls = int(row.class_id)\n\n#             # bỏ no finding\n#             if cls == 14:\n#                 continue\n\n#             x1 = row.x_min\n#             y1 = row.y_min\n#             x2 = row.x_max\n#             y2 = row.y_max\n\n#             xc = ((x1 + x2)/2) / w0\n#             yc = ((y1 + y2)/2) / h0\n\n#             bw = (x2 - x1) / w0\n#             bh = (y2 - y1) / h0\n\n#             lines.append(\n#                 f\"{cls} {xc} {yc} {bw} {bh}\"\n#             )\n\n#         label_out = (\n#             f\"{OUT_ROOT}/labels/\"\n#             f\"{split}/{image_id}.txt\"\n#         )\n\n#         with open(label_out, \"w\") as f:\n#             f.write(\"\\n\".join(lines))\n\n#     except Exception as e:\n\n#         print(image_id, e)\n\n# # =====================================================\n# # MULTIPROCESS\n# # =====================================================\n\n# with ProcessPoolExecutor(\n#     max_workers=NUM_WORKERS\n# ) as executor:\n\n#     list(\n#         tqdm(\n#             executor.map(\n#                 process_one,\n#                 image_ids,\n#                 chunksize=32\n#             ),\n#             total=len(image_ids)\n#         )\n#     )\n\n# print(\"DONE\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T11:49:55.89848Z","iopub.execute_input":"2026-05-12T11:49:55.902097Z","iopub.status.idle":"2026-05-12T12:40:37.134578Z","shell.execute_reply.started":"2026-05-12T11:49:55.902021Z","shell.execute_reply":"2026-05-12T12:40:37.131432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # preprocess xong\n\n# !tar -czf /kaggle/working/vinbig_yolo.tar.gz \\\n#     -C /kaggle/working vinbig_yolo\n\n# # xóa folder gốc\n# !rm -rf /kaggle/working/vinbig_yolo\n\n# print(\"DONE\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T12:40:37.142464Z","iopub.execute_input":"2026-05-12T12:40:37.143449Z","iopub.status.idle":"2026-05-12T12:41:43.453274Z","shell.execute_reply.started":"2026-05-12T12:40:37.1434Z","shell.execute_reply":"2026-05-12T12:41:43.451994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_yaml = \"\"\"\npath: /kaggle/input/datasets/keyshiftf/vinbig-yolo/vinbig_yolo\n\ntrain: images/train\nval: images/val\n\nnames:\n  0: Aortic enlargement\n  1: Atelectasis\n  2: Calcification\n  3: Cardiomegaly\n  4: Consolidation\n  5: ILD\n  6: Infiltration\n  7: Lung Opacity\n  8: Nodule/Mass\n  9: Other lesion\n  10: Pleural effusion\n  11: Pleural thickening\n  12: Pneumothorax\n  13: Pulmonary fibrosis\n\"\"\"\n\nwith open(\n    \"/kaggle/working/data.yaml\",\n    \"w\"\n) as f:\n\n    f.write(data_yaml)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T14:38:43.82544Z","iopub.execute_input":"2026-05-12T14:38:43.825837Z","iopub.status.idle":"2026-05-12T14:38:43.831247Z","shell.execute_reply.started":"2026-05-12T14:38:43.82581Z","shell.execute_reply":"2026-05-12T14:38:43.83045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ultralytics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T14:38:43.832306Z","iopub.execute_input":"2026-05-12T14:38:43.832561Z","iopub.status.idle":"2026-05-12T14:38:48.965635Z","shell.execute_reply.started":"2026-05-12T14:38:43.832537Z","shell.execute_reply":"2026-05-12T14:38:48.964527Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\nOptimized YOLO Training Script for Kaggle (T4x2 / P100)\nFeatures:\n- Multi-GPU support (T4x2)\n- Learning rate scheduling\n- Early stopping\n- Class balancing feedback\n- Memory optimization\n- Augmentation tuning\n\"\"\"\n\nfrom ultralytics import YOLO\nfrom pathlib import Path\nimport torch\nimport yaml\n\ndef detect_kaggle_gpu():\n    \"\"\"Detect available GPUs on Kaggle\"\"\"\n    if torch.cuda.is_available():\n        gpu_count = torch.cuda.device_count()\n        gpu_name = torch.cuda.get_device_name(0)\n        total_mem = torch.cuda.get_device_properties(0).total_memory / 1e9\n        print(f\"GPUs detected: {gpu_count}\")\n        print(f\"GPU Model: {gpu_name}\")\n        print(f\"Total Memory: {total_mem:.1f} GB\")\n        return gpu_count\n    else:\n        print(\"No GPU detected, using CPU (SLOW)\")\n        return 0\n\ndef analyze_dataset(data_yaml_path):\n    \"\"\"Analyze dataset class distribution\"\"\"\n    if not Path(data_yaml_path).exists():\n        print(f\"{data_yaml_path} not found, skipping analysis\")\n        return\n\n    with open(data_yaml_path) as f:\n        data = yaml.safe_load(f)\n\n    print(\"\\nDataset Configuration:\")\n    print(f\"  Classes: {data.get('nc', '?')}\")\n    print(f\"  Class names: {list(data.get('names', {}).values())}\")\n\ndef get_optimal_config(gpu_count, imgsz=1024):\n    \"\"\"Get optimal training config based on GPU\"\"\"\n\n    configs = {\n        \"t4_single\": {\n            \"batch\": 8,\n            \"workers\": 4,\n            \"imgsz\": 640,\n            \"accumulate\": 2,  # gradient accumulation\n            \"device\": 0,\n            \"note\": \"T4 (16GB) - Single GPU\"\n        },\n        \"t4_dual\": {\n            \"batch\": 16,\n            \"workers\": 8,\n            \"imgsz\": 768,\n            \"accumulate\": 1,\n            \"device\": [0, 1],\n            \"note\": \"T4x2 (32GB) - Dual GPU\"\n        },\n        \"p100\": {\n            \"batch\": 32,\n            \"workers\": 8,\n            \"imgsz\": 1024,\n            \"accumulate\": 1,\n            \"device\": 0,\n            \"note\": \"P100 (16GB) - High VRAM\"\n        }\n    }\n\n    # Auto-detect (heuristic)\n    if gpu_count >= 2:\n        return configs[\"t4_dual\"]\n    elif gpu_count == 1:\n        # Try to guess single T4 or P100 by memory\n        mem_gb = torch.cuda.get_device_properties(0).total_memory / 1e9\n        if mem_gb > 15:\n            return configs[\"p100\"]\n\n    return configs[\"t4_single\"]\n\ndef main():\n    # Step 1: Detect GPUs\n    print(\"=\" * 60)\n    print(\"YOLO Training on Kaggle - Optimized Setup\")\n    print(\"=\" * 60)\n\n    gpu_count = detect_kaggle_gpu()\n\n    # Step 2: Recommend config\n    config = get_optimal_config(gpu_count)\n    print(f\"\\n🔧 Recommended Config: {config['note']}\")\n\n    # Step 3: Analyze dataset\n    data_yaml = \"/kaggle/working/data.yaml\"\n    analyze_dataset(data_yaml)\n\n    # Step 4: Model selection\n    model_name = \"yolov8m.pt\"\n    if gpu_count >= 2:\n        print(f\"\\nTip: With dual T4, consider yolov8l.pt for better accuracy\")\n    else:\n        print(f\"\\nUsing {model_name} for balance between speed and accuracy\")\n\n    model = YOLO(model_name)\n\n    # Step 5: Training with optimizations\n    print(\"\\n\" + \"=\" * 60)\n    print(\"Starting Training with Advanced Features\")\n    print(\"=\" * 60)\n\n    train_config = {\n        # Data\n        \"data\": data_yaml,\n\n        # Training\n        \"epochs\": 50,\n        \"imgsz\": config[\"imgsz\"],\n        \"batch\": config[\"batch\"],\n        \"device\": config[\"device\"],\n        \"workers\": config[\"workers\"],\n\n        # Learning Rate Schedule (IMPORTANT!)\n        \"lr0\": 1e-4,                    # Initial LR\n        \"lrf\": 0.01,                    # Final LR fraction (1% of initial)\n        \"warmup_epochs\": 3.0,           # Gradual warmup\n        \"warmup_bias_lr\": 0.1,          # Warmup bias LR\n        \"warmup_momentum\": 0.8,         # Warmup momentum\n\n        # Optimizer & Weight Decay\n        \"optimizer\": \"AdamW\",\n        \"weight_decay\": 5e-4,\n        \"momentum\": 0.937,\n\n        # Early Stopping (IMPORTANT!)\n        \"patience\": 10,                 # Stop if no improvement for 10 epochs\n\n        # Performance\n        \"amp\": True,                    # Automatic Mixed Precision (faster)\n        \"cache\": \"ram\",                 # Cache images in RAM (faster training)\n\n        # Augmentation (tuned for Kaggle resources)\n        \"fliplr\": 0.5,\n        \"flipud\": 0.1,\n        \"degrees\": 10,                  # Increased for better robustness\n        \"translate\": 0.1,               # Increased\n        \"scale\": 0.2,                   # Increased\n        \"mosaic\": 1.0,                  # Enable (was disabled, enable it!)\n        \"mixup\": 0.2,                   # Enable for better generalization\n        \"hsv_h\": 0.015,                 # HSV hue\n        \"hsv_s\": 0.7,                   # HSV saturation\n        \"hsv_v\": 0.4,                   # HSV value\n        \"erasing\": 0.4,                 # Random erasing\n        \"dropout\": 0.0,\n\n        # Validation\n        \"val\": True,\n        \"save\": True,\n        \"plots\": True,\n\n        # Checkpointing\n        \"save_period\": 10,              # Save checkpoint every 10 epochs\n        \"exist_ok\": True,\n\n        # Reproducibility\n        \"seed\": 42,\n        \"deterministic\": True,\n\n        # Logging\n        \"verbose\": True,\n    }\n\n    # Add gradient accumulation for smaller effective batch size\n    if config[\"accumulate\"] > 1:\n        train_config[\"accumulate\"] = config[\"accumulate\"]\n\n    print(\"\\nTraining Configuration:\")\n    print(f\"  Model: {model_name}\")\n    print(f\"  Batch Size: {train_config['batch']}\")\n    print(f\"  Image Size: {train_config['imgsz']}\")\n    print(f\"  Learning Rate: {train_config['lr0']} → {train_config['lrf']*train_config['lr0']}\")\n    print(f\"  Warmup: {train_config['warmup_epochs']} epochs\")\n    print(f\"  Early Stopping: {train_config['patience']} epochs patience\")\n    print(f\"  Augmentation: Mosaic={train_config['mosaic']}, Mixup={train_config['mixup']}\")\n    print(f\"  Device(s): {train_config['device']}\")\n\n    # Train\n    results = model.train(**train_config)\n\n    # Print results summary\n    print(\"\\n\" + \"=\" * 60)\n    print(\"Training Complete!\")\n    print(\"=\" * 60)\n    print(f\"Best Model Saved: runs/detect/train/weights/best.pt\")\n    print(f\"Metrics Saved: runs/detect/train/results.csv\")\n\n    return results\n\nif __name__ == \"__main__\":\n    results = main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T14:38:48.967191Z","iopub.execute_input":"2026-05-12T14:38:48.967544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\n\nmodel = YOLO(\n    \"/kaggle/working/runs/detect/train/weights/best.pt\"\n)\n\nmetrics = model.val()\n\nprint(metrics)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T14:07:26.747799Z","iopub.status.idle":"2026-05-12T14:07:26.749276Z","shell.execute_reply.started":"2026-05-12T14:07:26.749025Z","shell.execute_reply":"2026-05-12T14:07:26.74906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\n\nmodel = YOLO(\n    \"/kaggle/working/runs/detect/train/weights/best.pt\"\n)\n\nmetrics = model.val(\n    plots=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T14:07:26.750319Z","iopub.status.idle":"2026-05-12T14:07:26.750746Z","shell.execute_reply.started":"2026-05-12T14:07:26.750511Z","shell.execute_reply":"2026-05-12T14:07:26.75056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}