{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"kernelVersion","sourceId":317348452},{"sourceType":"kernelVersion","sourceId":317995557}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q ultralytics\n!pip install -q pandas numpy opencv-python tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T05:09:34.06022Z","iopub.execute_input":"2026-05-10T05:09:34.060864Z","iopub.status.idle":"2026-05-10T05:09:41.440375Z","shell.execute_reply.started":"2026-05-10T05:09:34.060822Z","shell.execute_reply":"2026-05-10T05:09:41.439276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ndef list_files(startpath):\n    print(f\"--- Kiểm tra cấu trúc thư mục tại: {startpath} ---\")\n    if not os.path.exists(startpath):\n        print(\"❌ Đường dẫn không tồn tại!\")\n        return\n        \n    for root, dirs, files in os.walk(startpath):\n        level = root.replace(startpath, '').count(os.sep)\n        indent = ' ' * 4 * (level)\n        print(f\"{indent}📂 {os.path.basename(root)}/\")\n        subindent = ' ' * 4 * (level + 1)\n        for f in files:\n            print(f\"{subindent}📄 {f}\")\n\n# Đường dẫn bạn muốn kiểm tra\npath_to_check = '/kaggle/input/notebooks/nguyenphanthanhan/notebook-prep-d-t'\nlist_files(path_to_check)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T05:09:41.442684Z","iopub.execute_input":"2026-05-10T05:09:41.443105Z","iopub.status.idle":"2026-05-10T05:10:18.548291Z","shell.execute_reply.started":"2026-05-10T05:09:41.443057Z","shell.execute_reply":"2026-05-10T05:10:18.547594Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport shutil\nfrom tqdm.auto import tqdm\n\n# --- 1. Cấu hình đường dẫn ---\nINPUT_DIR = '/kaggle/input/notebooks/nguyenphanthanhan/notebook-prep-d-t'\nIMG_DIR = os.path.join(INPUT_DIR, 'train_images_1024')\nRAW_CSV_PATH = os.path.join(INPUT_DIR, 'train_1024.csv')\nCLS_DATA_DIR = '/kaggle/working/classifier_data'\n\n# --- 2. Xử lý nhãn nhị phân ---\ndf_raw = pd.read_csv(RAW_CSV_PATH)\nabnormal_ids = set(df_raw[df_raw['class_id'] != 14]['image_id'].unique())\nall_image_ids = [f.replace('.png', '') for f in os.listdir(IMG_DIR) if f.endswith('.png')]\n\n# --- 3. Dọn dẹp và chuẩn bị thư mục (Cần dọn sạch 17GB cũ) ---\nif os.path.exists(CLS_DATA_DIR):\n    shutil.rmtree(CLS_DATA_DIR)\n\nfor split in ['train', 'val']:\n    for label in ['normal', 'abnormal']:\n        os.makedirs(os.path.join(CLS_DATA_DIR, split, label), exist_ok=True)\n\n# --- 4. Chia tập Train/Val ---\nnp.random.seed(42)\nval_ids = set(np.random.choice(all_image_ids, int(len(all_image_ids) * 0.2), replace=False))\n\n# --- 5. Tạo Symlinks (Thay vì copy/xử lý ảnh) ---\nprint(\"--- Đang tạo liên kết (Symlinks) ---\")\nfor img_id in tqdm(all_image_ids):\n    target = 1 if img_id in abnormal_ids else 0\n    split = 'val' if img_id in val_ids else 'train'\n    label_name = 'abnormal' if target == 1 else 'normal'\n    \n    src_path = os.path.join(IMG_DIR, f\"{img_id}.png\")\n    dst_path = os.path.join(CLS_DATA_DIR, split, label_name, f\"{img_id}.png\")\n    \n    # Tạo symlink - tốn 0 bytes\n    if os.path.exists(src_path):\n        os.symlink(src_path, dst_path)\n\n# --- 6. Xuất thêm file CSV để bạn theo dõi (Theo ý bạn) ---\ndf_mapping = pd.DataFrame({\n    'image_id': all_image_ids,\n    'label': [1 if x in abnormal_ids else 0 for x in all_image_ids],\n    'split': ['val' if x in val_ids else 'train' for x in all_image_ids]\n})\ndf_mapping.to_csv('/kaggle/working/metadata_mapping.csv', index=False)\n\nprint(f\"✅ Hoàn tất! Dung lượng ổ đĩa sử dụng sẽ giảm đáng kể.\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T05:10:18.549334Z","iopub.execute_input":"2026-05-10T05:10:18.549869Z","iopub.status.idle":"2026-05-10T05:10:31.254028Z","shell.execute_reply.started":"2026-05-10T05:10:18.549844Z","shell.execute_reply":"2026-05-10T05:10:31.252938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# [Cell 11 - Tối ưu tham số] Huấn luyện Phân loại có Checkpoint\nfrom ultralytics import YOLO\n\n# Khởi tạo mô hình phân loại bản Medium\nmodel = YOLO('yolov8m-cls.pt')\n\n# Bắt đầu huấn luyện\nresults = model.train(\n    data='/kaggle/working/classifier_data',\n    epochs=30,\n    imgsz=1024,\n    batch=16,                # Giảm xuống 16 hoặc đặt batch=-1 để dùng AutoBatch\n    workers=4,               # Phân luồng nạp dữ liệu từ CPU (Kaggle thường có 4 nhân)\n    device=0,\n    optimizer='AdamW',\n    lr0=0.001,\n    patience=10,\n    # --- CẤU HÌNH CHECKPOINT ---\n    save=True,               \n    save_period=5,           \n    project='/kaggle/working/training_results',\n    name='stage1_classifier_binary'\n)\n\nprint(\"✅ Huấn luyện Stage 1 hoàn tất!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T05:10:31.255776Z","iopub.execute_input":"2026-05-10T05:10:31.256135Z"}},"outputs":[],"execution_count":null}]}