{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"kernelVersion","sourceId":317348452}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q ultralytics\n!pip install -q pandas numpy opencv-python tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T03:57:48.977678Z","iopub.execute_input":"2026-05-10T03:57:48.978013Z","iopub.status.idle":"2026-05-10T03:58:00.474259Z","shell.execute_reply.started":"2026-05-10T03:57:48.977982Z","shell.execute_reply":"2026-05-10T03:58:00.472715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport cv2\nimport shutil\nfrom tqdm.auto import tqdm\nfrom concurrent.futures import ProcessPoolExecutor, as_completed\n\n# --- 1. Cấu hình đường dẫn ---\nINPUT_DIR = '/kaggle/input/notebooks/nguyenphanthanhan/notebook-prep-d-t'\nIMG_DIR = os.path.join(INPUT_DIR, 'train_images_1024')\nRAW_CSV_PATH = os.path.join(INPUT_DIR, 'train_1024.csv')\nCLS_DATA_DIR = '/kaggle/working/classifier_data'\n\n# --- 2. Xử lý nhãn nhị phân chuẩn xác ---\nprint(\"--- Đang quét toàn bộ dữ liệu (Target ~15,000 images) ---\")\ndf_raw = pd.read_csv(RAW_CSV_PATH)\nabnormal_ids = set(df_raw[df_raw['class_id'] != 14]['image_id'].unique())\n\nall_image_files = [f for f in os.listdir(IMG_DIR) if f.endswith('.png')]\nall_image_ids = [f.replace('.png', '') for f in all_image_files]\n\ndf_binary = pd.DataFrame({'image_id': all_image_ids})\ndf_binary['target'] = df_binary['image_id'].apply(lambda x: 1 if x in abnormal_ids else 0)\n\nprint(f\"Tổng số ảnh quét được: {len(df_binary)}\")\nprint(f\"Phân bổ nhãn: {df_binary['target'].value_counts().to_dict()}\") \n\n# --- 3. Chuẩn bị thư mục ---\nif os.path.exists(CLS_DATA_DIR):\n    shutil.rmtree(CLS_DATA_DIR)\n\nfor split in ['train', 'val']:\n    for label in ['normal', 'abnormal']:\n        os.makedirs(os.path.join(CLS_DATA_DIR, split, label), exist_ok=True)\n\n# --- 4. Chia tập Train/Val (80/20) ---\nnp.random.seed(42)\nval_ids = set(np.random.choice(all_image_ids, int(len(all_image_ids) * 0.2), replace=False))\n\n# --- 5. Hàm xử lý đơn lẻ để chạy Đa luồng ---\ndef process_single_image(img_id, target, val_ids, img_dir, cls_data_dir):\n    src_path = os.path.join(img_dir, f\"{img_id}.png\")\n    if not os.path.exists(src_path):\n        return False\n        \n    split = 'val' if img_id in val_ids else 'train'\n    label_name = 'abnormal' if target == 1 else 'normal'\n    dst_path = os.path.join(cls_data_dir, split, label_name, f\"{img_id}.png\")\n\n    img = cv2.imread(src_path, cv2.IMREAD_GRAYSCALE)\n    if img is not None:\n        final_img = cv2.cvtColor(img, cv2.COLOR_GRAY2BGR)\n        cv2.imwrite(dst_path, final_img)\n        return True\n    return False\n\n# --- 6. Thực thi Đa luồng (Tối ưu tốc độ) ---\nprint(\"--- Bắt đầu xử lý đa luồng ---\")\n# Sử dụng ProcessPoolExecutor để tận dụng tối đa các nhân CPU của Kaggle\nworkers = os.cpu_count() or 4 \n\nwith ProcessPoolExecutor(max_workers=workers) as executor:\n    # Gửi tất cả các task vào pool\n    futures = [\n        executor.submit(process_single_image, row['image_id'], row['target'], val_ids, IMG_DIR, CLS_DATA_DIR)\n        for _, row in df_binary.iterrows()\n    ]\n    \n    # Hiển thị thanh tiến trình\n    for future in tqdm(as_completed(futures), total=len(futures), desc=\"Đang xử lý 15k ảnh\"):\n        pass # Task đã hoàn thành\n\nprint(f\"✅ Đã chuẩn bị xong {len(df_binary)} ảnh tại: {CLS_DATA_DIR}\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T04:07:06.365684Z","iopub.execute_input":"2026-05-10T04:07:06.366302Z","iopub.status.idle":"2026-05-10T04:18:00.378242Z","shell.execute_reply.started":"2026-05-10T04:07:06.366251Z","shell.execute_reply":"2026-05-10T04:18:00.376795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # [Cell 11 - Cập nhật] Huấn luyện Phân loại có Checkpoint\n# from ultralytics import YOLO\n\n# # Khởi tạo mô hình phân loại bản Medium\n# model = YOLO('yolov8m-cls.pt')\n\n# # Bắt đầu huấn luyện\n# results = model.train(\n#     data='/kaggle/working/classifier_data',\n#     epochs=30,               \n#     imgsz=1024,               \n#     batch=32,                \n#     device=0,                \n#     optimizer='AdamW',       \n#     lr0=0.001,\n#     patience=10,             \n    \n#     # --- CẤU HÌNH CHECKPOINT ---\n#     save=True,               # Bật tính năng lưu tệp trọng số (mặc định là True)\n#     save_period=5,           # LƯU CHECKPOINT MỖI 5 EPOCH (ví dụ: epoch5.pt, epoch10.pt...)\n    \n#     project='/kaggle/working/training_results',\n#     name='stage1_classifier_binary'\n# )\n\n# print(\"✅ Huấn luyện Stage 1 hoàn tất!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T03:27:14.172709Z","iopub.execute_input":"2026-05-06T03:27:14.173019Z","iopub.status.idle":"2026-05-06T05:28:32.982156Z","shell.execute_reply.started":"2026-05-06T03:27:14.172995Z","shell.execute_reply":"2026-05-06T05:28:32.976779Z"}},"outputs":[],"execution_count":null}]}