{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\ninput_dir = Path(\"/kaggle/input\")\n\nfor p in input_dir.glob(\"*\"):\n    print(p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-01T19:42:54.859732Z","iopub.execute_input":"2026-07-01T19:42:54.859977Z","iopub.status.idle":"2026-07-01T19:42:54.867713Z","shell.execute_reply.started":"2026-07-01T19:42:54.859955Z","shell.execute_reply":"2026-07-01T19:42:54.867061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nbase = Path(\"/kaggle/input/datasets\")\n\nfor p in base.glob(\"**/*\"):\n    print(p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-01T23:22:13.989432Z","iopub.execute_input":"2026-07-01T23:22:13.990192Z","iopub.status.idle":"2026-07-01T23:22:24.08155Z","shell.execute_reply.started":"2026-07-01T23:22:13.990163Z","shell.execute_reply":"2026-07-01T23:22:24.080695Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nbase = Path(\"/kaggle/input/datasets/xhlulu/vinbigdata\")\n\ncsv_files = list(base.glob(\"*.csv\"))\n\nfor p in csv_files:\n    print(p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-01T23:22:44.094737Z","iopub.execute_input":"2026-07-01T23:22:44.095424Z","iopub.status.idle":"2026-07-01T23:22:44.10112Z","shell.execute_reply.started":"2026-07-01T23:22:44.095396Z","shell.execute_reply":"2026-07-01T23:22:44.100253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nfor p in Path(\"/kaggle/input\").glob(\"**/train.csv\"):\n    print(p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-01T23:22:48.600702Z","iopub.execute_input":"2026-07-01T23:22:48.601004Z","iopub.status.idle":"2026-07-01T23:23:18.689696Z","shell.execute_reply.started":"2026-07-01T23:22:48.60098Z","shell.execute_reply":"2026-07-01T23:23:18.689079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport pandas as pd\nfrom PIL import Image\n\n# Thư mục ảnh PNG 512x512\nPNG_BASE = Path(\"/kaggle/input/datasets/xhlulu/vinbigdata\")\nIMG_DIR = PNG_BASE / \"train\"\nMETA_PATH = PNG_BASE / \"train_meta.csv\"\n\n# Tìm file train.csv từ dataset gốc\ntrain_csv_files = list(Path(\"/kaggle/input\").glob(\"**/train.csv\"))\n\nprint(\"Các file train.csv tìm thấy:\")\nfor p in train_csv_files:\n    print(p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-01T23:24:08.679113Z","iopub.execute_input":"2026-07-01T23:24:08.679608Z","iopub.status.idle":"2026-07-01T23:24:16.957851Z","shell.execute_reply.started":"2026-07-01T23:24:08.679581Z","shell.execute_reply":"2026-07-01T23:24:16.957133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ANN_PATH = train_csv_files[0]\n\ndf = pd.read_csv(ANN_PATH)\nmeta = pd.read_csv(META_PATH)\n\nprint(\"ANN_PATH:\", ANN_PATH)\nprint(\"META_PATH:\", META_PATH)\nprint(\"IMG_DIR:\", IMG_DIR)\n\nprint(\"\\nKích thước train.csv:\", df.shape)\nprint(\"Kích thước train_meta.csv:\", meta.shape)\n\nprint(\"\\nCác cột trong train.csv:\")\nprint(df.columns.tolist())\n\nprint(\"\\nCác cột trong train_meta.csv:\")\nprint(meta.columns.tolist())\n\nprint(\"\\n5 dòng đầu của train.csv:\")\ndisplay(df.head())\n\nprint(\"\\n5 dòng đầu của train_meta.csv:\")\ndisplay(meta.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-01T23:24:26.733844Z","iopub.execute_input":"2026-07-01T23:24:26.73453Z","iopub.status.idle":"2026-07-01T23:24:26.844879Z","shell.execute_reply.started":"2026-07-01T23:24:26.734504Z","shell.execute_reply":"2026-07-01T23:24:26.84428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_DIR = Path(\"/kaggle/input/datasets/xhlulu/vinbigdata/train\")\n\n# Lấy thử 1 ảnh có bệnh, không lấy No finding\nsample_id = df[df[\"class_id\"] != 14][\"image_id\"].iloc[0]\n\nsample_img = IMG_DIR / f\"{sample_id}.png\"\n\nprint(\"Sample image_id:\", sample_id)\nprint(\"Đường dẫn ảnh:\", sample_img)\nprint(\"Ảnh có tồn tại không?\", sample_img.exists())\n\nif sample_img.exists():\n    img = Image.open(sample_img)\n    print(\"Kích thước ảnh PNG:\", img.size)\n    display(img)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-01T23:24:31.513523Z","iopub.execute_input":"2026-07-01T23:24:31.514232Z","iopub.status.idle":"2026-07-01T23:24:31.587402Z","shell.execute_reply.started":"2026-07-01T23:24:31.514205Z","shell.execute_reply":"2026-07-01T23:24:31.586559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom PIL import Image, ImageDraw, ImageFont\nfrom pathlib import Path\n\n# Merge train.csv với train_meta.csv để có dim0, dim1\ndf_meta = df.merge(meta, on=\"image_id\", how=\"left\")\n\n# Lấy sample_id từ ảnh có bệnh ở bước trước\nsample_id = df[df[\"class_id\"] != 14][\"image_id\"].iloc[0]\n\nimg_path = IMG_DIR / f\"{sample_id}.png\"\nimg = Image.open(img_path).convert(\"RGB\")\n\ndraw = ImageDraw.Draw(img)\n\n# Lấy tất cả bbox của ảnh này\nboxes = df_meta[\n    (df_meta[\"image_id\"] == sample_id) &\n    (df_meta[\"class_id\"] != 14)\n]\n\nprint(\"image_id:\", sample_id)\nprint(\"Số bbox:\", len(boxes))\ndisplay(boxes)\n\n# Kích thước ảnh PNG sau resize\nnew_w, new_h = img.size\n\nfor _, row in boxes.iterrows():\n    # Kích thước ảnh gốc\n    old_h = row[\"dim0\"]\n    old_w = row[\"dim1\"]\n\n    # Scale bbox từ ảnh gốc về ảnh PNG 512x512\n    x_min = row[\"x_min\"] * new_w / old_w\n    x_max = row[\"x_max\"] * new_w / old_w\n    y_min = row[\"y_min\"] * new_h / old_h\n    y_max = row[\"y_max\"] * new_h / old_h\n\n    class_name = row[\"class_name\"]\n\n    # Vẽ khung\n    draw.rectangle(\n        [x_min, y_min, x_max, y_max],\n        outline=\"red\",\n        width=3\n    )\n\n    # Ghi tên bệnh\n    draw.text(\n        (x_min, max(0, y_min - 12)),\n        class_name,\n        fill=\"red\"\n    )\n\nplt.figure(figsize=(8, 8))\nplt.imshow(img, cmap=\"gray\")\nplt.axis(\"off\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-01T23:24:38.21821Z","iopub.execute_input":"2026-07-01T23:24:38.218753Z","iopub.status.idle":"2026-07-01T23:24:38.455531Z","shell.execute_reply.started":"2026-07-01T23:24:38.218726Z","shell.execute_reply":"2026-07-01T23:24:38.454644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Gộp train.csv với train_meta.csv để có dim0, dim1\ndf_meta = df.merge(meta, on=\"image_id\", how=\"left\")\n\n# Chỉ lấy ảnh có bounding box, bỏ class 14 = No finding\nbbox_df = df_meta[\n    (df_meta[\"class_id\"] != 14) &\n    df_meta[[\"x_min\", \"y_min\", \"x_max\", \"y_max\", \"dim0\", \"dim1\"]].notna().all(axis=1)\n].copy()\n\n# Đảm bảo class_id là số nguyên\nbbox_df[\"class_id\"] = bbox_df[\"class_id\"].astype(int)\n\n# Chuyển bbox từ dạng VinBigData sang YOLO\nbbox_df[\"x_center\"] = ((bbox_df[\"x_min\"] + bbox_df[\"x_max\"]) / 2) / bbox_df[\"dim1\"]\nbbox_df[\"y_center\"] = ((bbox_df[\"y_min\"] + bbox_df[\"y_max\"]) / 2) / bbox_df[\"dim0\"]\n\nbbox_df[\"box_width\"] = (bbox_df[\"x_max\"] - bbox_df[\"x_min\"]) / bbox_df[\"dim1\"]\nbbox_df[\"box_height\"] = (bbox_df[\"y_max\"] - bbox_df[\"y_min\"]) / bbox_df[\"dim0\"]\n\n# Chặn giá trị về khoảng 0 đến 1\nfor col in [\"x_center\", \"y_center\", \"box_width\", \"box_height\"]:\n    bbox_df[col] = bbox_df[col].clip(0, 1)\n\n# Kiểm tra bbox lỗi\ninvalid_boxes = bbox_df[\n    (bbox_df[\"box_width\"] <= 0) |\n    (bbox_df[\"box_height\"] <= 0) |\n    (bbox_df[\"x_center\"] < 0) |\n    (bbox_df[\"x_center\"] > 1) |\n    (bbox_df[\"y_center\"] < 0) |\n    (bbox_df[\"y_center\"] > 1)\n]\n\nprint(\"Số dòng bbox hợp lệ:\", len(bbox_df))\nprint(\"Số ảnh có bbox:\", bbox_df[\"image_id\"].nunique())\nprint(\"Số dòng bbox lỗi:\", len(invalid_boxes))\n\ndisplay(bbox_df[[\n    \"image_id\", \"class_name\", \"class_id\",\n    \"x_min\", \"y_min\", \"x_max\", \"y_max\",\n    \"dim0\", \"dim1\",\n    \"x_center\", \"y_center\", \"box_width\", \"box_height\"\n]].head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-01T23:38:28.464721Z","iopub.execute_input":"2026-07-01T23:38:28.465503Z","iopub.status.idle":"2026-07-01T23:38:28.530588Z","shell.execute_reply.started":"2026-07-01T23:38:28.465472Z","shell.execute_reply":"2026-07-01T23:38:28.52992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nfrom sklearn.model_selection import train_test_split\n\n# Thư mục YOLO sẽ được tạo trong /kaggle/working\nYOLO_DIR = Path(\"/kaggle/working/vinbigdata_yolo\")\n\n# Tạo cấu trúc thư mục\nfor split in [\"train\", \"val\"]:  #8:2\n    (YOLO_DIR / \"images\" / split).mkdir(parents=True, exist_ok=True)\n    (YOLO_DIR / \"labels\" / split).mkdir(parents=True, exist_ok=True)\n\n# Lấy danh sách image_id có bbox\nimage_ids = bbox_df[\"image_id\"].unique()\n\n# Chia train/val theo tỉ lệ 80/20\ntrain_ids, val_ids = train_test_split(\n    image_ids,\n    test_size=0.2,\n    random_state=42\n)\n\nprint(\"YOLO_DIR:\", YOLO_DIR)\nprint(\"Tổng số ảnh có bbox:\", len(image_ids))\nprint(\"Số ảnh train:\", len(train_ids))\nprint(\"Số ảnh val:\", len(val_ids))\n\nprint(\"\\nKiểm tra thư mục đã tạo:\")\nfor p in [\n    YOLO_DIR / \"images\" / \"train\",\n    YOLO_DIR / \"images\" / \"val\",\n    YOLO_DIR / \"labels\" / \"train\",\n    YOLO_DIR / \"labels\" / \"val\",\n]:\n    print(p, \"->\", p.exists())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-01T23:41:40.455222Z","iopub.execute_input":"2026-07-01T23:41:40.456104Z","iopub.status.idle":"2026-07-01T23:41:41.25118Z","shell.execute_reply.started":"2026-07-01T23:41:40.456073Z","shell.execute_reply":"2026-07-01T23:41:41.250434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nfrom tqdm import tqdm\n\n# IMG_DIR là thư mục ảnh PNG 512x512\nIMG_DIR = Path(\"/kaggle/input/datasets/xhlulu/vinbigdata/train\")\n\ndef link_or_copy_image(src_path, dst_path):\n    \"\"\"\n    Ưu tiên tạo symlink để tiết kiệm dung lượng.\n    Nếu symlink lỗi thì copy ảnh.\n    \"\"\"\n    if dst_path.exists():\n        return\n    \n    try:\n        os.symlink(src_path, dst_path)\n    except:\n        shutil.copy2(src_path, dst_path)\n\n\ndef create_yolo_files(image_id, split):\n    # Đường dẫn ảnh gốc PNG\n    src_img = IMG_DIR / f\"{image_id}.png\"\n    \n    if not src_img.exists():\n        return False\n    \n    # Đường dẫn ảnh trong thư mục YOLO\n    dst_img = YOLO_DIR / \"images\" / split / f\"{image_id}.png\"\n    \n    # Đường dẫn label YOLO\n    label_path = YOLO_DIR / \"labels\" / split / f\"{image_id}.txt\"\n    \n    # Lấy tất cả bbox của ảnh này\n    rows = bbox_df[bbox_df[\"image_id\"] == image_id]\n    \n    lines = []\n    \n    for _, row in rows.iterrows():\n        class_id = int(row[\"class_id\"])\n        x_center = row[\"x_center\"]\n        y_center = row[\"y_center\"]\n        box_width = row[\"box_width\"]\n        box_height = row[\"box_height\"]\n        \n        line = f\"{class_id} {x_center:.6f} {y_center:.6f} {box_width:.6f} {box_height:.6f}\"\n        lines.append(line)\n    \n    # Ghi file label .txt\n    with open(label_path, \"w\") as f:\n        f.write(\"\\n\".join(lines))\n    \n    # Link/copy ảnh\n    link_or_copy_image(src_img, dst_img)\n    \n    return True\n\n\n# Tạo dữ liệu train\ntrain_success = 0\nfor image_id in tqdm(train_ids, desc=\"Creating train YOLO files\"):\n    if create_yolo_files(image_id, \"train\"):\n        train_success += 1\n\n# Tạo dữ liệu val\nval_success = 0\nfor image_id in tqdm(val_ids, desc=\"Creating val YOLO files\"):\n    if create_yolo_files(image_id, \"val\"):\n        val_success += 1\n\n\nprint(\"Số ảnh train tạo thành công:\", train_success)\nprint(\"Số ảnh val tạo thành công:\", val_success)\n\nprint(\"Số file ảnh train:\", len(list((YOLO_DIR / \"images\" / \"train\").glob(\"*.png\"))))\nprint(\"Số file label train:\", len(list((YOLO_DIR / \"labels\" / \"train\").glob(\"*.txt\"))))\n\nprint(\"Số file ảnh val:\", len(list((YOLO_DIR / \"images\" / \"val\").glob(\"*.png\"))))\nprint(\"Số file label val:\", len(list((YOLO_DIR / \"labels\" / \"val\").glob(\"*.txt\"))))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-01T23:43:55.12989Z","iopub.execute_input":"2026-07-01T23:43:55.130802Z","iopub.status.idle":"2026-07-01T23:44:14.069301Z","shell.execute_reply.started":"2026-07-01T23:43:55.130773Z","shell.execute_reply":"2026-07-01T23:44:14.068516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Kiểm thử xem ảnh và label bbox có đúng không\nimport random\nfrom PIL import Image, ImageDraw\nimport matplotlib.pyplot as plt\n\n# Lấy ngẫu nhiên 1 ảnh trong tập train\nsample_img_path = random.choice(list((YOLO_DIR / \"images\" / \"train\").glob(\"*.png\")))\nsample_label_path = YOLO_DIR / \"labels\" / \"train\" / f\"{sample_img_path.stem}.txt\"\n\nprint(\"Ảnh:\", sample_img_path)\nprint(\"Label:\", sample_label_path)\n\n# In nội dung file label YOLO\nwith open(sample_label_path, \"r\") as f:\n    label_lines = f.readlines()\n\nprint(\"\\nNội dung label YOLO:\")\nfor line in label_lines:\n    print(line.strip())\n\n# Mở ảnh\nimg = Image.open(sample_img_path).convert(\"RGB\")\ndraw = ImageDraw.Draw(img)\n\nimg_w, img_h = img.size\n\n# Danh sách tên class\nnames = [\n    \"Aortic enlargement\",\n    \"Atelectasis\",\n    \"Calcification\",\n    \"Cardiomegaly\",\n    \"Consolidation\",\n    \"ILD\",\n    \"Infiltration\",\n    \"Lung Opacity\",\n    \"Nodule/Mass\",\n    \"Other lesion\",\n    \"Pleural effusion\",\n    \"Pleural thickening\",\n    \"Pneumothorax\",\n    \"Pulmonary fibrosis\"\n]\n\n# Vẽ lại bbox từ format YOLO\nfor line in label_lines:\n    parts = line.strip().split()\n    \n    class_id = int(parts[0])\n    x_center = float(parts[1])\n    y_center = float(parts[2])\n    box_w = float(parts[3])\n    box_h = float(parts[4])\n    \n    # Đổi từ YOLO normalized về pixel trên ảnh 512x512\n    x_center *= img_w\n    y_center *= img_h\n    box_w *= img_w\n    box_h *= img_h\n    \n    x_min = x_center - box_w / 2\n    y_min = y_center - box_h / 2\n    x_max = x_center + box_w / 2\n    y_max = y_center + box_h / 2\n    \n    draw.rectangle(\n        [x_min, y_min, x_max, y_max],\n        outline=\"red\",\n        width=3\n    )\n    \n    draw.text(\n        (x_min, max(0, y_min - 12)),\n        names[class_id],\n        fill=\"red\"\n    )\n\nplt.figure(figsize=(8, 8))\nplt.imshow(img)\nplt.axis(\"off\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-01T23:45:39.34085Z","iopub.execute_input":"2026-07-01T23:45:39.341241Z","iopub.status.idle":"2026-07-01T23:45:39.534174Z","shell.execute_reply.started":"2026-07-01T23:45:39.341212Z","shell.execute_reply":"2026-07-01T23:45:39.533038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Danh sách 14 class bệnh của VinBigData, bỏ class 14 = No finding\nnames = [\n    \"Aortic enlargement\",\n    \"Atelectasis\",\n    \"Calcification\",\n    \"Cardiomegaly\",\n    \"Consolidation\",\n    \"ILD\",\n    \"Infiltration\",\n    \"Lung Opacity\",\n    \"Nodule/Mass\",\n    \"Other lesion\",\n    \"Pleural effusion\",\n    \"Pleural thickening\",\n    \"Pneumothorax\",\n    \"Pulmonary fibrosis\"\n]\n\nyaml_text = f\"\"\"\npath: {YOLO_DIR}\ntrain: images/train\nval: images/val\n\nnc: 14\nnames:\n\"\"\"\n\nfor i, name in enumerate(names):\n    yaml_text += f\"  {i}: {name}\\n\"\n\nyaml_path = YOLO_DIR / \"dataset.yaml\"\n\nwith open(yaml_path, \"w\") as f:\n    f.write(yaml_text)\n\nprint(\"Đã tạo file dataset.yaml tại:\")\nprint(yaml_path)\n\nprint(\"\\nNội dung dataset.yaml:\")\nprint(yaml_text)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-01T23:48:25.230255Z","iopub.execute_input":"2026-07-01T23:48:25.230791Z","iopub.status.idle":"2026-07-01T23:48:25.236948Z","shell.execute_reply.started":"2026-07-01T23:48:25.230762Z","shell.execute_reply":"2026-07-01T23:48:25.236319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cài YOLO\n!pip install ultralytics -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-01T23:50:36.744601Z","iopub.execute_input":"2026-07-01T23:50:36.744952Z","iopub.status.idle":"2026-07-01T23:50:42.961506Z","shell.execute_reply.started":"2026-07-01T23:50:36.744926Z","shell.execute_reply":"2026-07-01T23:50:42.96081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#ultralytics hỗ trơtrợ YOLOv8,YOLO11 \nimport torch\nimport ultralytics\n\nprint(\"CUDA available:\", torch.cuda.is_available())\n\nif torch.cuda.is_available():\n    print(\"GPU name:\", torch.cuda.get_device_name(0))\n\nultralytics.checks()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-01T23:52:38.901321Z","iopub.execute_input":"2026-07-01T23:52:38.902334Z","iopub.status.idle":"2026-07-01T23:52:43.36844Z","shell.execute_reply.started":"2026-07-01T23:52:38.902293Z","shell.execute_reply":"2026-07-01T23:52:43.367849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom ultralytics import YOLO\n\n# Tắt wandb để tránh hỏi login\nos.environ[\"WANDB_DISABLED\"] = \"true\"\n\n# Dùng YOLOv8n: bản nhẹ nhất, phù hợp để test trước\nmodel = YOLO(\"yolov8n.pt\")\n\nresults = model.train(\n    data=str(yaml_path),\n    epochs=5,\n    imgsz=512,\n    batch=8,\n    device=0,\n    workers=2,\n    project=\"/kaggle/working/runs\",\n    name=\"vinbigdata_yolov8n_test\",\n    exist_ok=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-01T23:53:50.296112Z","iopub.execute_input":"2026-07-01T23:53:50.297002Z","iopub.status.idle":"2026-07-01T23:54:04.69487Z","shell.execute_reply.started":"2026-07-01T23:53:50.296969Z","shell.execute_reply":"2026-07-01T23:54:04.693531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\n\nprint(\"CUDA available:\", torch.cuda.is_available())\n\nif torch.cuda.is_available():\n    print(\"GPU count:\", torch.cuda.device_count())\n    for i in range(torch.cuda.device_count()):\n        print(i, torch.cuda.get_device_name(i))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-01T23:57:22.166233Z","iopub.execute_input":"2026-07-01T23:57:22.166472Z","iopub.status.idle":"2026-07-01T23:57:27.786384Z","shell.execute_reply.started":"2026-07-01T23:57:22.166447Z","shell.execute_reply":"2026-07-01T23:57:27.785184Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport torch\n\nprint(\"CUDA available:\", torch.cuda.is_available())\n\nif torch.cuda.is_available():\n    print(\"GPU count:\", torch.cuda.device_count())\n    for i in range(torch.cuda.device_count()):\n        print(i, torch.cuda.get_device_name(i))\n\n# Khôi phục lại đường dẫn YOLO sau khi restart\nYOLO_DIR = Path(\"/kaggle/working/vinbigdata_yolo\")\nyaml_path = YOLO_DIR / \"dataset.yaml\"\n\nprint(\"\\nYOLO_DIR tồn tại không?\", YOLO_DIR.exists())\nprint(\"dataset.yaml tồn tại không?\", yaml_path.exists())\n\nprint(\"\\nSố ảnh train:\", len(list((YOLO_DIR / \"images\" / \"train\").glob(\"*.png\"))) if YOLO_DIR.exists() else 0)\nprint(\"Số label train:\", len(list((YOLO_DIR / \"labels\" / \"train\").glob(\"*.txt\"))) if YOLO_DIR.exists() else 0)\n\nprint(\"Số ảnh val:\", len(list((YOLO_DIR / \"images\" / \"val\").glob(\"*.png\"))) if YOLO_DIR.exists() else 0)\nprint(\"Số label val:\", len(list((YOLO_DIR / \"labels\" / \"val\").glob(\"*.txt\"))) if YOLO_DIR.exists() else 0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-02T00:00:29.306368Z","iopub.execute_input":"2026-07-02T00:00:29.307415Z","iopub.status.idle":"2026-07-02T00:00:29.316557Z","shell.execute_reply.started":"2026-07-02T00:00:29.307383Z","shell.execute_reply":"2026-07-02T00:00:29.315642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport pandas as pd\nimport os\nimport shutil\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\n\n# =========================\n# 1. Khai báo đường dẫn\n# =========================\n\nPNG_BASE = Path(\"/kaggle/input/datasets/xhlulu/vinbigdata\")\nIMG_DIR = PNG_BASE / \"train\"\nMETA_PATH = PNG_BASE / \"train_meta.csv\"\n\n# Tìm file train.csv từ dataset gốc VinBigData\ntrain_csv_files = list(Path(\"/kaggle/input\").glob(\"**/train.csv\"))\n\nprint(\"Các file train.csv tìm thấy:\")\nfor p in train_csv_files:\n    print(p)\n\nANN_PATH = train_csv_files[0]\n\nprint(\"\\nDùng ANN_PATH:\", ANN_PATH)\nprint(\"Dùng META_PATH:\", META_PATH)\nprint(\"Dùng IMG_DIR:\", IMG_DIR)\n\n# =========================\n# 2. Đọc dữ liệu\n# =========================\n\ndf = pd.read_csv(ANN_PATH)\nmeta = pd.read_csv(META_PATH)\n\ndf_meta = df.merge(meta, on=\"image_id\", how=\"left\")\n\n# Bỏ class 14 = No finding, chỉ giữ bbox bệnh\nbbox_df = df_meta[\n    (df_meta[\"class_id\"] != 14) &\n    df_meta[[\"x_min\", \"y_min\", \"x_max\", \"y_max\", \"dim0\", \"dim1\"]].notna().all(axis=1)\n].copy()\n\nbbox_df[\"class_id\"] = bbox_df[\"class_id\"].astype(int)\n\n# =========================\n# 3. Chuyển bbox sang YOLO format\n# =========================\n\nbbox_df[\"x_center\"] = ((bbox_df[\"x_min\"] + bbox_df[\"x_max\"]) / 2) / bbox_df[\"dim1\"]\nbbox_df[\"y_center\"] = ((bbox_df[\"y_min\"] + bbox_df[\"y_max\"]) / 2) / bbox_df[\"dim0\"]\n\nbbox_df[\"box_width\"] = (bbox_df[\"x_max\"] - bbox_df[\"x_min\"]) / bbox_df[\"dim1\"]\nbbox_df[\"box_height\"] = (bbox_df[\"y_max\"] - bbox_df[\"y_min\"]) / bbox_df[\"dim0\"]\n\nfor col in [\"x_center\", \"y_center\", \"box_width\", \"box_height\"]:\n    bbox_df[col] = bbox_df[col].clip(0, 1)\n\ninvalid_boxes = bbox_df[\n    (bbox_df[\"box_width\"] <= 0) |\n    (bbox_df[\"box_height\"] <= 0)\n]\n\nprint(\"\\nSố dòng bbox:\", len(bbox_df))\nprint(\"Số ảnh có bbox:\", bbox_df[\"image_id\"].nunique())\nprint(\"Số bbox lỗi:\", len(invalid_boxes))\n\n# =========================\n# 4. Tạo thư mục YOLO\n# =========================\n\nYOLO_DIR = Path(\"/kaggle/working/vinbigdata_yolo\")\n\nif YOLO_DIR.exists():\n    shutil.rmtree(YOLO_DIR)\n\nfor split in [\"train\", \"val\"]:\n    (YOLO_DIR / \"images\" / split).mkdir(parents=True, exist_ok=True)\n    (YOLO_DIR / \"labels\" / split).mkdir(parents=True, exist_ok=True)\n\n# =========================\n# 5. Chia train / val\n# =========================\n\nimage_ids = bbox_df[\"image_id\"].unique()\n\ntrain_ids, val_ids = train_test_split(\n    image_ids,\n    test_size=0.2,\n    random_state=42\n)\n\nprint(\"\\nSố ảnh train:\", len(train_ids))\nprint(\"Số ảnh val:\", len(val_ids))\n\n# =========================\n# 6. Tạo ảnh và label YOLO\n# =========================\n\ndef link_or_copy_image(src_path, dst_path):\n    if dst_path.exists():\n        return\n    \n    try:\n        os.symlink(src_path, dst_path)\n    except:\n        shutil.copy2(src_path, dst_path)\n\n\ndef create_yolo_files(image_id, split):\n    src_img = IMG_DIR / f\"{image_id}.png\"\n    \n    if not src_img.exists():\n        return False\n    \n    dst_img = YOLO_DIR / \"images\" / split / f\"{image_id}.png\"\n    label_path = YOLO_DIR / \"labels\" / split / f\"{image_id}.txt\"\n    \n    rows = bbox_df[bbox_df[\"image_id\"] == image_id]\n    \n    lines = []\n    \n    for _, row in rows.iterrows():\n        class_id = int(row[\"class_id\"])\n        x_center = row[\"x_center\"]\n        y_center = row[\"y_center\"]\n        box_width = row[\"box_width\"]\n        box_height = row[\"box_height\"]\n        \n        lines.append(\n            f\"{class_id} {x_center:.6f} {y_center:.6f} {box_width:.6f} {box_height:.6f}\"\n        )\n    \n    with open(label_path, \"w\") as f:\n        f.write(\"\\n\".join(lines))\n    \n    link_or_copy_image(src_img, dst_img)\n    \n    return True\n\n\ntrain_success = 0\nfor image_id in tqdm(train_ids, desc=\"Creating train YOLO files\"):\n    if create_yolo_files(image_id, \"train\"):\n        train_success += 1\n\nval_success = 0\nfor image_id in tqdm(val_ids, desc=\"Creating val YOLO files\"):\n    if create_yolo_files(image_id, \"val\"):\n        val_success += 1\n\n# =========================\n# 7. Tạo dataset.yaml\n# =========================\n\nnames = [\n    \"Aortic enlargement\",\n    \"Atelectasis\",\n    \"Calcification\",\n    \"Cardiomegaly\",\n    \"Consolidation\",\n    \"ILD\",\n    \"Infiltration\",\n    \"Lung Opacity\",\n    \"Nodule/Mass\",\n    \"Other lesion\",\n    \"Pleural effusion\",\n    \"Pleural thickening\",\n    \"Pneumothorax\",\n    \"Pulmonary fibrosis\"\n]\n\nyaml_text = f\"\"\"\npath: {YOLO_DIR}\ntrain: images/train\nval: images/val\n\nnc: 14\nnames:\n\"\"\"\n\nfor i, name in enumerate(names):\n    yaml_text += f\"  {i}: {name}\\n\"\n\nyaml_path = YOLO_DIR / \"dataset.yaml\"\n\nwith open(yaml_path, \"w\") as f:\n    f.write(yaml_text)\n\n# =========================\n# 8. Kiểm tra kết quả\n# =========================\n\nprint(\"\\nTạo YOLO dataset xong.\")\nprint(\"YOLO_DIR:\", YOLO_DIR)\nprint(\"yaml_path:\", yaml_path)\n\nprint(\"\\nSố ảnh train:\", len(list((YOLO_DIR / \"images\" / \"train\").glob(\"*.png\"))))\nprint(\"Số label train:\", len(list((YOLO_DIR / \"labels\" / \"train\").glob(\"*.txt\"))))\n\nprint(\"Số ảnh val:\", len(list((YOLO_DIR / \"images\" / \"val\").glob(\"*.png\"))))\nprint(\"Số label val:\", len(list((YOLO_DIR / \"labels\" / \"val\").glob(\"*.txt\"))))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-02T00:01:43.664569Z","iopub.execute_input":"2026-07-02T00:01:43.664888Z","iopub.status.idle":"2026-07-02T00:02:43.790998Z","shell.execute_reply.started":"2026-07-02T00:01:43.664839Z","shell.execute_reply":"2026-07-02T00:02:43.790172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ultralytics -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-02T00:05:04.805557Z","iopub.execute_input":"2026-07-02T00:05:04.806191Z","iopub.status.idle":"2026-07-02T00:05:11.730462Z","shell.execute_reply.started":"2026-07-02T00:05:04.806158Z","shell.execute_reply":"2026-07-02T00:05:11.729192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom ultralytics import YOLO\n\n# Tắt wandb để tránh hỏi login\nos.environ[\"WANDB_DISABLED\"] = \"true\"\n\nYOLO_DIR = \"/kaggle/working/vinbigdata_yolo\"\nyaml_path = \"/kaggle/working/vinbigdata_yolo/dataset.yaml\"\n\n# YOLOv8n = bản nhẹ nhất, train thử cho chắc pipeline\nmodel = YOLO(\"yolov8n.pt\")\n\nresults = model.train(\n    data=yaml_path,\n    epochs=5,\n    imgsz=512,\n    batch=8,\n    device=0,          # dùng GPU T4 số 0 thôi cho ổn định\n    workers=2,\n    project=\"/kaggle/working/runs\",\n    name=\"vinbigdata_yolov8n_test\",\n    exist_ok=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-02T00:05:16.18561Z","iopub.execute_input":"2026-07-02T00:05:16.186685Z","iopub.status.idle":"2026-07-02T00:10:17.315928Z","shell.execute_reply.started":"2026-07-02T00:05:16.186641Z","shell.execute_reply":"2026-07-02T00:10:17.315021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nfrom IPython.display import Image, display\n\nRUN_DIR = Path(\"/kaggle/working/runs/vinbigdata_yolov8n_test\")\n\nprint(\"Các file trong thư mục kết quả:\")\nfor p in RUN_DIR.glob(\"*\"):\n    print(p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-02T00:11:46.176743Z","iopub.execute_input":"2026-07-02T00:11:46.177432Z","iopub.status.idle":"2026-07-02T00:11:46.185484Z","shell.execute_reply.started":"2026-07-02T00:11:46.177384Z","shell.execute_reply":"2026-07-02T00:11:46.184652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results_img = RUN_DIR / \"results.png\"\n\nprint(\"results.png tồn tại không?\", results_img.exists())\n\nif results_img.exists():\n    display(Image(filename=str(results_img)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-02T00:12:02.566375Z","iopub.execute_input":"2026-07-02T00:12:02.566712Z","iopub.status.idle":"2026-07-02T00:12:02.587007Z","shell.execute_reply.started":"2026-07-02T00:12:02.566683Z","shell.execute_reply":"2026-07-02T00:12:02.585941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_images = list(RUN_DIR.glob(\"val_batch*_pred.jpg\"))\n\nprint(\"Số ảnh dự đoán validation:\", len(pred_images))\n\nfor p in pred_images[:3]:\n    print(p)\n    display(Image(filename=str(p)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-02T00:12:22.572333Z","iopub.execute_input":"2026-07-02T00:12:22.572648Z","iopub.status.idle":"2026-07-02T00:12:22.612936Z","shell.execute_reply.started":"2026-07-02T00:12:22.572619Z","shell.execute_reply":"2026-07-02T00:12:22.612219Z"}},"outputs":[],"execution_count":null}]}