{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":14491631,"datasetId":9255756,"databundleVersionId":15316389},{"sourceType":"datasetVersion","sourceId":15621315,"datasetId":9997952,"databundleVersionId":16555619},{"sourceType":"datasetVersion","sourceId":15359032,"datasetId":9824215,"databundleVersionId":16269544},{"sourceType":"datasetVersion","sourceId":14609699,"datasetId":9331890,"databundleVersionId":15446019},{"sourceType":"datasetVersion","sourceId":14620660,"datasetId":9338847,"databundleVersionId":15457983},{"sourceType":"datasetVersion","sourceId":15183452,"datasetId":9342173,"databundleVersionId":16076564},{"sourceType":"datasetVersion","sourceId":15358993,"datasetId":9824192,"databundleVersionId":16269503},{"sourceType":"datasetVersion","sourceId":15501878,"datasetId":9917891,"databundleVersionId":16427513},{"sourceType":"datasetVersion","sourceId":1799839,"datasetId":1069682,"databundleVersionId":1837296},{"sourceType":"datasetVersion","sourceId":1800067,"datasetId":1069810,"databundleVersionId":1837524},{"sourceType":"datasetVersion","sourceId":1800777,"datasetId":1069787,"databundleVersionId":1838236},{"sourceType":"datasetVersion","sourceId":15609727,"datasetId":9989282,"databundleVersionId":16543384}],"dockerImageVersionId":31234,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":22636.689771,"end_time":"2026-01-13T22:48:25.795437","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-01-13T16:31:09.105666","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!nvidia-smi\n","metadata":{"execution":{"iopub.status.busy":"2026-05-17T03:48:51.441036Z","iopub.execute_input":"2026-05-17T03:48:51.441762Z","iopub.status.idle":"2026-05-17T03:48:51.9082Z","shell.execute_reply.started":"2026-05-17T03:48:51.441729Z","shell.execute_reply":"2026-05-17T03:48:51.907467Z"},"papermill":{"duration":0.25722,"end_time":"2026-01-13T16:31:12.908207","exception":false,"start_time":"2026-01-13T16:31:12.650987","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ultralytics==8.4.36 --quiet","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2026-05-17T03:48:51.909989Z","iopub.execute_input":"2026-05-17T03:48:51.910202Z","iopub.status.idle":"2026-05-17T03:48:57.425604Z","shell.execute_reply.started":"2026-05-17T03:48:51.910179Z","shell.execute_reply":"2026-05-17T03:48:57.424801Z"},"papermill":{"duration":6.651321,"end_time":"2026-01-13T16:31:19.562598","exception":false,"start_time":"2026-01-13T16:31:12.911277","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import ultralytics\nprint(ultralytics.__version__)  # phải ra 8.4.36","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-17T03:48:57.427025Z","iopub.execute_input":"2026-05-17T03:48:57.427329Z","iopub.status.idle":"2026-05-17T03:49:01.158425Z","shell.execute_reply.started":"2026-05-17T03:48:57.427296Z","shell.execute_reply":"2026-05-17T03:49:01.157855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport numpy as np\nimport shutil\nimport yaml\nimport matplotlib.pyplot as plt\nimport random\nimport cv2\n\nfrom sklearn import model_selection\nfrom tqdm import tqdm\nfrom glob import glob","metadata":{"execution":{"iopub.status.busy":"2026-05-17T03:49:01.160269Z","iopub.execute_input":"2026-05-17T03:49:01.160677Z","iopub.status.idle":"2026-05-17T03:49:02.128729Z","shell.execute_reply.started":"2026-05-17T03:49:01.160605Z","shell.execute_reply":"2026-05-17T03:49:02.128128Z"},"papermill":{"duration":4.462325,"end_time":"2026-01-13T16:31:24.028715","exception":false,"start_time":"2026-01-13T16:31:19.56639","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"size = 1024\nTRAIN_LABELS_PATH = './vinbigdata/labels/train'\nVAL_LABELS_PATH = './vinbigdata/labels/val'\nTRAIN_IMAGES_PATH = './vinbigdata/images/train' #12000\nVAL_IMAGES_PATH = './vinbigdata/images/val' #3000\nExternal_DIR = f'../input/vinbigdata-{size}-image-dataset/vinbigdata/train' # 15000\nos.makedirs(TRAIN_LABELS_PATH, exist_ok = True)\nos.makedirs(VAL_LABELS_PATH, exist_ok = True)\nos.makedirs(TRAIN_IMAGES_PATH, exist_ok = True)\nos.makedirs(VAL_IMAGES_PATH, exist_ok = True)","metadata":{"execution":{"iopub.status.busy":"2026-05-17T03:49:02.129588Z","iopub.execute_input":"2026-05-17T03:49:02.130034Z","iopub.status.idle":"2026-05-17T03:49:02.137109Z","shell.execute_reply.started":"2026-05-17T03:49:02.129989Z","shell.execute_reply":"2026-05-17T03:49:02.136262Z"},"papermill":{"duration":0.010756,"end_time":"2026-01-13T16:31:24.042726","exception":false,"start_time":"2026-01-13T16:31:24.03197","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"original_df = pd.read_csv('../input/vinbigdata-chest-xray-abnormalities-detection/train.csv')\nnumber_of_imageids = len(original_df['image_id'].values)\nprint(f'Total number of image_ids (train + validation) {number_of_imageids}')\n\nnumber_of_images = len(os.listdir('../input/vinbigdata-chest-xray-abnormalities-detection/train'))\nprint(f'Total number of images (train + validation) {number_of_images}')\n\nnumber_of_labels = len(os.listdir('../input/vinbigdata-yolo-labels-dataset/labels'))\nprint(f'Total number of labels (train + validation) {number_of_labels}')","metadata":{"execution":{"iopub.status.busy":"2026-05-17T03:49:02.13804Z","iopub.execute_input":"2026-05-17T03:49:02.138331Z","iopub.status.idle":"2026-05-17T03:49:03.051482Z","shell.execute_reply.started":"2026-05-17T03:49:02.138289Z","shell.execute_reply":"2026-05-17T03:49:03.05084Z"},"papermill":{"duration":0.640246,"end_time":"2026-01-13T16:31:24.686241","exception":false,"start_time":"2026-01-13T16:31:24.045995","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv')\nnumber_of_images = len(df['image_id'].values)\nprint(f'Total number of image ids (train + validation) {number_of_images}')\n\ndf = df[df.class_id!=14].reset_index(drop = True)\nnumber_of_images = len(df['image_id'].values)\nprint(f'Total number of image ids after dropping normal images (train + validation) {number_of_images}')\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2026-05-17T03:49:03.052471Z","iopub.execute_input":"2026-05-17T03:49:03.052838Z","iopub.status.idle":"2026-05-17T03:49:03.164278Z","shell.execute_reply.started":"2026-05-17T03:49:03.052812Z","shell.execute_reply":"2026-05-17T03:49:03.163388Z"},"papermill":{"duration":0.122103,"end_time":"2026-01-13T16:31:24.812155","exception":false,"start_time":"2026-01-13T16:31:24.690052","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.drop(columns=['class_name', 'rad_id', 'x_min', 'x_max', 'y_min', 'y_max',  'class_id']) # we only need image ids, labels are pre-made\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2026-05-17T03:49:03.16522Z","iopub.execute_input":"2026-05-17T03:49:03.165488Z","iopub.status.idle":"2026-05-17T03:49:03.175309Z","shell.execute_reply.started":"2026-05-17T03:49:03.165463Z","shell.execute_reply":"2026-05-17T03:49:03.174679Z"},"papermill":{"duration":0.01788,"end_time":"2026-01-13T16:31:24.833714","exception":false,"start_time":"2026-01-13T16:31:24.815834","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train, df_valid = model_selection.train_test_split(df, test_size=0.15, random_state=42, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2026-05-17T03:49:03.176288Z","iopub.execute_input":"2026-05-17T03:49:03.176567Z","iopub.status.idle":"2026-05-17T03:49:03.189768Z","shell.execute_reply.started":"2026-05-17T03:49:03.176543Z","shell.execute_reply":"2026-05-17T03:49:03.189079Z"},"papermill":{"duration":0.01257,"end_time":"2026-01-13T16:31:24.849944","exception":false,"start_time":"2026-01-13T16:31:24.837374","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"number_of_images = len(df_train['image_id'].values)\nprint(f'Total number of training image_ids {number_of_images}')\n\nnumber_of_images = len(df_valid['image_id'].values)\nprint(f'Total number of validation image_ids {number_of_images}')\n","metadata":{"execution":{"iopub.status.busy":"2026-05-17T03:49:03.192392Z","iopub.execute_input":"2026-05-17T03:49:03.192702Z","iopub.status.idle":"2026-05-17T03:49:03.199811Z","shell.execute_reply.started":"2026-05-17T03:49:03.192668Z","shell.execute_reply":"2026-05-17T03:49:03.199109Z"},"papermill":{"duration":0.011329,"end_time":"2026-01-13T16:31:24.866919","exception":false,"start_time":"2026-01-13T16:31:24.85559","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'Total number of training images {len(df_train.image_id.unique())}')\nprint(f'Total number of validation images {len(df_valid.image_id.unique())}')","metadata":{"execution":{"iopub.status.busy":"2026-05-17T03:49:03.200729Z","iopub.execute_input":"2026-05-17T03:49:03.201001Z","iopub.status.idle":"2026-05-17T03:49:03.21625Z","shell.execute_reply.started":"2026-05-17T03:49:03.200973Z","shell.execute_reply":"2026-05-17T03:49:03.215539Z"},"papermill":{"duration":0.016563,"end_time":"2026-01-13T16:31:24.887187","exception":false,"start_time":"2026-01-13T16:31:24.870624","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preproccess_data(df, labels_path, images_path):\n    for img_id in tqdm(df.image_id.unique()):\n        shutil.copy(os.path.join('../input/vinbigdata-yolo-labels-dataset/labels', f\"{img_id}\"+'.txt'), labels_path)\n        shutil.copy(os.path.join(f'/kaggle/input/vinbigdata-{size}-image-dataset/vinbigdata/train', f\"{img_id}.png\"), images_path)","metadata":{"execution":{"iopub.status.busy":"2026-05-17T03:49:03.217224Z","iopub.execute_input":"2026-05-17T03:49:03.217779Z","iopub.status.idle":"2026-05-17T03:49:03.224604Z","shell.execute_reply.started":"2026-05-17T03:49:03.21774Z","shell.execute_reply":"2026-05-17T03:49:03.223978Z"},"papermill":{"duration":0.010627,"end_time":"2026-01-13T16:31:24.9017","exception":false,"start_time":"2026-01-13T16:31:24.891073","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preproccess_data(df_train, TRAIN_LABELS_PATH, TRAIN_IMAGES_PATH)\npreproccess_data(df_valid, VAL_LABELS_PATH, VAL_IMAGES_PATH)","metadata":{"execution":{"iopub.status.busy":"2026-05-17T03:49:03.225654Z","iopub.execute_input":"2026-05-17T03:49:03.225943Z","iopub.status.idle":"2026-05-17T03:50:44.789854Z","shell.execute_reply.started":"2026-05-17T03:49:03.225919Z","shell.execute_reply":"2026-05-17T03:50:44.789002Z"},"papermill":{"duration":67.99992,"end_time":"2026-01-13T16:32:32.905626","exception":false,"start_time":"2026-01-13T16:31:24.905706","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check that data was preprocessed correctly\nprint(len(os.listdir(TRAIN_LABELS_PATH)))\nprint(len(os.listdir(TRAIN_IMAGES_PATH)))\n\nprint(len(os.listdir(VAL_LABELS_PATH)))\nprint(len(os.listdir(VAL_IMAGES_PATH)))","metadata":{"execution":{"iopub.status.busy":"2026-05-17T03:50:44.790821Z","iopub.execute_input":"2026-05-17T03:50:44.791036Z","iopub.status.idle":"2026-05-17T03:50:44.805789Z","shell.execute_reply.started":"2026-05-17T03:50:44.791015Z","shell.execute_reply":"2026-05-17T03:50:44.804962Z"},"papermill":{"duration":0.040904,"end_time":"2026-01-13T16:32:32.973019","exception":false,"start_time":"2026-01-13T16:32:32.932115","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"classes = [ 'Aortic enlargement',\n            'Atelectasis',\n            'Calcification',\n            'Cardiomegaly',\n            'Consolidation',\n            'ILD',\n            'Infiltration',\n            'Lung Opacity',\n            'Nodule/Mass',\n            'Other lesion',\n            'Pleural effusion',\n            'Pleural thickening',\n            'Pneumothorax',\n            'Pulmonary fibrosis']\n\ndata = dict(\n    train =  '../vinbigdata/images/train',\n    val   =  '../vinbigdata/images/val',\n    nc    = 14,\n    names = classes\n    )\n\nwith open('/kaggle/working/vinbigdata.yaml', 'w') as outfile:\n    yaml.dump(data, outfile, default_flow_style=False)\n\nf = open(os.path.join( os.getcwd() , 'vinbigdata.yaml'), 'r')\nprint('\\nyaml:')\nprint(f.read())","metadata":{"execution":{"iopub.status.busy":"2026-05-17T03:50:44.806834Z","iopub.execute_input":"2026-05-17T03:50:44.807212Z","iopub.status.idle":"2026-05-17T03:50:44.814101Z","shell.execute_reply.started":"2026-05-17T03:50:44.807176Z","shell.execute_reply":"2026-05-17T03:50:44.813471Z"},"papermill":{"duration":0.034917,"end_time":"2026-01-13T16:32:33.033847","exception":false,"start_time":"2026-01-13T16:32:32.99893","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# import torch\n# from ultralytics import YOLO\n\n# os.environ[\"WANDB_MODE\"] = \"dryrun\"\n# assert torch.cuda.is_available(), \"⚠️ GPU NOT AVAILABLE. Check Kaggle Accelerator.\"\n\n\n# model = YOLO(\"yolo11l.pt\")\n\n# results = model.train(\n#     data='./vinbigdata.yaml',\n\n#     imgsz=1024,\n#     batch=8,\n#     epochs=120,\n#     patience=25,\n\n#     optimizer=\"AdamW\",\n#     lr0=0.0015,\n#     lrf=0.01,\n#     weight_decay=5e-4,\n#     cos_lr=True,\n\n#     hsv_h=0.01,\n#     hsv_s=0.15,  \n#     hsv_v=0.20,\n#     degrees=3.0,\n#     translate=0.1,\n#     scale=0.2,\n#     flipud=0.0,\n#     fliplr=0.5,\n#     mosaic=0.5,\n#     mixup=0.1,\n#     close_mosaic=20,\n\n#     box=8.0,\n#     cls=3.0,\n#     dfl=2.0,\n#     iou=0.55,\n\n#     device=0,\n#     workers=2,\n#     amp=True,\n#     cache=\"ram\",\n\n#     project=\"/kaggle/working/runs\",\n#     name=\"yolo11l_highprecision_v3\",\n#     save=True,\n#     save_period=1,\n#     exist_ok=True,\n#     resume=False,\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-17T03:50:44.815026Z","iopub.execute_input":"2026-05-17T03:50:44.815299Z","iopub.status.idle":"2026-05-17T03:50:44.826247Z","shell.execute_reply.started":"2026-05-17T03:50:44.815275Z","shell.execute_reply":"2026-05-17T03:50:44.825444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# import torch\n# from ultralytics import YOLO\n\n# os.environ[\"WANDB_MODE\"] = \"dryrun\"\n# assert torch.cuda.is_available(), \"⚠️ GPU NOT AVAILABLE. Check Kaggle Accelerator.\"\n\n\n# model = YOLO(\"/kaggle/input/datasets/ggduck14/version21/last.pt\")\n\n# results = model.train(\n#     data='./vinbigdata.yaml',\n\n#     imgsz=1024,\n#     batch=8,\n#     epochs=120,\n#     patience=25,\n\n#     optimizer=\"AdamW\",\n#     lr0=0.0015,\n#     lrf=0.01,\n#     weight_decay=5e-4,\n#     cos_lr=True,\n\n#     hsv_h=0.01,\n#     hsv_s=0.15,  \n#     hsv_v=0.20,\n#     degrees=3.0,\n#     translate=0.1,\n#     scale=0.2,\n#     flipud=0.0,\n#     fliplr=0.5,\n#     mosaic=0.3,\n#     mixup=0.0,\n#     close_mosaic=20,\n\n#     box=8.0,\n#     cls=3.0,\n#     dfl=2.0,\n#     iou=0.55,\n\n#     device=0,\n#     workers=2,\n#     amp=True,\n#     cache=\"ram\",\n\n#     project=\"/kaggle/working/runs\",\n#     name=\"yolo11l_highprecision_v3_final\",\n#     save=True,\n#     save_period=1,\n#     exist_ok=True,\n#     resume=True,\n# )","metadata":{"execution":{"iopub.status.busy":"2026-05-17T03:50:44.827148Z","iopub.execute_input":"2026-05-17T03:50:44.827757Z","iopub.status.idle":"2026-05-17T03:50:44.838727Z","shell.execute_reply.started":"2026-05-17T03:50:44.827733Z","shell.execute_reply":"2026-05-17T03:50:44.838132Z"},"papermill":{"duration":22107.832563,"end_time":"2026-01-13T22:41:00.893812","exception":false,"start_time":"2026-01-13T16:32:33.061249","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# import torch\n# from ultralytics import YOLO\n\n# os.environ[\"WANDB_MODE\"] = \"dryrun\"\n# assert torch.cuda.is_available(), \"⚠️ GPU NOT AVAILABLE. Check Kaggle Accelerator.\"\n\n\n# model = YOLO(\"yolov8l.pt\")\n\n# results = model.train(\n#     data='./vinbigdata.yaml',\n\n#     imgsz=1024,\n#     batch=8,\n#     epochs=120,\n#     patience=25,\n\n#     optimizer=\"AdamW\",\n#     lr0=0.0015,\n#     lrf=0.01,\n#     weight_decay=5e-4,\n#     cos_lr=True,\n\n#     hsv_h=0.01,\n#     hsv_s=0.15,  \n#     hsv_v=0.20,\n#     degrees=3.0,\n#     translate=0.1,\n#     scale=0.2,\n#     flipud=0.0,\n#     fliplr=0.5,\n#     mosaic=0.3,\n#     mixup=0.0,\n#     close_mosaic=20,\n\n#     box=8.0,\n#     cls=3.0,\n#     dfl=2.0,\n#     iou=0.55,\n\n#     device=0,\n#     workers=2,\n#     amp=True,\n#     cache=\"ram\",\n\n#     project=\"/kaggle/working/runs\",\n#     name=\"yolo11l_highprecision_v4\",\n#     save=True,\n#     save_period=1,\n#     exist_ok=True,\n#     resume=True,\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-17T03:50:44.839594Z","iopub.execute_input":"2026-05-17T03:50:44.839921Z","iopub.status.idle":"2026-05-17T03:50:44.850427Z","shell.execute_reply.started":"2026-05-17T03:50:44.839899Z","shell.execute_reply":"2026-05-17T03:50:44.849652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# import torch\n# from ultralytics import YOLO\n\n# os.environ[\"WANDB_MODE\"] = \"dryrun\"\n# assert torch.cuda.is_available(), \"⚠️ GPU NOT AVAILABLE. Check Kaggle Accelerator.\"\n\n\n# model = YOLO(\"/kaggle/input/datasets/ggduck14/yolov8/lastYolov8.pt\")\n\n# results = model.train(\n#     data='./vinbigdata.yaml',\n\n#     imgsz=1024,\n#     batch=8,\n#     epochs=120,\n#     patience=25,\n\n#     optimizer=\"AdamW\",\n#     lr0=0.0015,\n#     lrf=0.01,\n#     weight_decay=5e-4,\n#     cos_lr=True,\n\n#     hsv_h=0.01,\n#     hsv_s=0.15,  \n#     hsv_v=0.20,\n#     degrees=3.0,\n#     translate=0.1,\n#     scale=0.2,\n#     flipud=0.0,\n#     fliplr=0.5,\n#     mosaic=0.3,\n#     mixup=0.0,\n#     close_mosaic=20,\n\n#     box=8.0,\n#     cls=3.0,\n#     dfl=2.0,\n#     iou=0.55,\n\n#     device=0,\n#     workers=2,\n#     amp=True,\n#     cache=\"ram\",\n\n#     project=\"/kaggle/working/runs\",\n#     name=\"yolo11l_highprecision_v4_final\",\n#     save=True,\n#     exist_ok=True,\n#     resume=True,\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-17T03:50:44.851179Z","iopub.execute_input":"2026-05-17T03:50:44.851388Z","iopub.status.idle":"2026-05-17T03:50:44.86968Z","shell.execute_reply.started":"2026-05-17T03:50:44.851367Z","shell.execute_reply":"2026-05-17T03:50:44.868847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from ultralytics import YOLO\n# model = YOLO(\"/kaggle/input/version-16/best.pt\")\n\n# r = model.val(data=\"./vinbigdata.yaml\", imgsz=640, device=\"cpu\", plots=False, conf=0.325, iou=0.7)\n# print(r.box.map50, r.box.map)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-17T03:50:44.870652Z","iopub.execute_input":"2026-05-17T03:50:44.871385Z","iopub.status.idle":"2026-05-17T03:50:44.881334Z","shell.execute_reply.started":"2026-05-17T03:50:44.871348Z","shell.execute_reply":"2026-05-17T03:50:44.880501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !zip -r /kaggle/working/yolo_all_output.zip /kaggle/working\n","metadata":{"execution":{"iopub.status.busy":"2026-05-17T03:50:44.882191Z","iopub.execute_input":"2026-05-17T03:50:44.882474Z","iopub.status.idle":"2026-05-17T03:50:44.897013Z","shell.execute_reply.started":"2026-05-17T03:50:44.882452Z","shell.execute_reply":"2026-05-17T03:50:44.896215Z"},"papermill":{"duration":135.283454,"end_time":"2026-01-13T22:48:20.997518","exception":false,"start_time":"2026-01-13T22:46:05.714064","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\n\nmodel = YOLO(\"/kaggle/input/version-16/best.pt\")\n\nfor conf in [0.05, 0.1, 0.15, 0.2, 0.25, 0.3, 0.35, 0.4, 0.5]:\n    metrics = model.val(\n        data=\"/kaggle/working/vinbigdata.yaml\",\n        imgsz=1024,\n        split=\"val\",\n        conf=conf,\n        verbose=False\n    )\n    r = metrics.results_dict\n    print(\n        f\"conf={conf:.2f} | \"\n        f\"P={r['metrics/precision(B)']:.4f} | \"\n        f\"R={r['metrics/recall(B)']:.4f} | \"\n        f\"mAP50={r['metrics/mAP50(B)']:.4f} | \"\n        f\"mAP50-95={r['metrics/mAP50-95(B)']:.4f}\"\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-17T03:50:44.897911Z","iopub.execute_input":"2026-05-17T03:50:44.898177Z","iopub.status.idle":"2026-05-17T03:50:44.907218Z","shell.execute_reply.started":"2026-05-17T03:50:44.898142Z","shell.execute_reply":"2026-05-17T03:50:44.906645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from ultralytics import YOLO\n\n# model = YOLO(\"/kaggle/input/version-16/best.pt\")\n\n# metrics = model.val(\n#     data=\"/kaggle/working/vinbigdata.yaml\",\n#     imgsz=1024,\n#     split=\"val\",\n#     conf=0.25,\n#     iou=0.5\n# )\n\n# print(metrics.results_dict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-17T03:50:44.907967Z","iopub.execute_input":"2026-05-17T03:50:44.908234Z","iopub.status.idle":"2026-05-17T03:50:44.918569Z","shell.execute_reply.started":"2026-05-17T03:50:44.908199Z","shell.execute_reply":"2026-05-17T03:50:44.917943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from ultralytics import YOLO\n\n# model = YOLO(\"/kaggle/input/version-16/best.pt\")\n\n# for conf in [0.05, 0.1, 0.15, 0.2, 0.25, 0.3, 0.4, 0.5, 0.6]:\n#     metrics = model.val(\n#         data=\"/kaggle/working/vinbigdata.yaml\",\n#         imgsz=1024,\n#         split=\"val\",\n#         conf=conf,\n#         iou=0.65,\n#         verbose=False\n#     )\n#     r = metrics.results_dict\n#     print(\n#         r\n#     )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-17T03:50:44.919495Z","iopub.execute_input":"2026-05-17T03:50:44.919789Z","iopub.status.idle":"2026-05-17T03:50:44.930327Z","shell.execute_reply.started":"2026-05-17T03:50:44.919754Z","shell.execute_reply":"2026-05-17T03:50:44.929517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from ultralytics import YOLO\n\n# model = YOLO(\"/kaggle/input/version-16\")\n\n# for conf in [0.05, 0.1, 0.15, 0.2, 0.25, 0.3, 0.4, 0.5, 0.6]:\n#     metrics = model.val(\n#         data=\"/kaggle/working/vinbigdata.yaml\",\n#         imgsz=1024,\n#         split=\"val\",\n#         conf=conf,\n#         iou=0.65,\n#         verbose=False\n#     )\n#     r = metrics.results_dict\n#     print(\n#         f\"conf={conf:.2f} | \"\n#         f\"P={r['metrics/precision(B)']:.4f} | \"\n#         f\"R={r['metrics/recall(B)']:.4f} | \"\n#         f\"mAP50={r['metrics/mAP50(B)']:.4f} | \"\n#         f\"mAP50-95={r['metrics/mAP50-95(B)']:.4f}\"\n#     )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-17T03:50:44.931213Z","iopub.execute_input":"2026-05-17T03:50:44.931487Z","iopub.status.idle":"2026-05-17T03:50:44.942041Z","shell.execute_reply.started":"2026-05-17T03:50:44.931456Z","shell.execute_reply":"2026-05-17T03:50:44.941375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\nimport pandas as pd\n\n# Load model\nmodel = YOLO(\"/kaggle/input/version-16/best.pt\")\n\n# Run validation\nmetrics = model.val(\n    data=\"vinbigdata.yaml\",\n    imgsz=1024,\n    split=\"val\",\n    conf=0.20,     \n    iou=0.55,      \n    verbose=False\n)\n\n\n# =========================\n# 1. OVERALL METRICS\n# =========================\nprint(\"\\n===== OVERALL METRICS YOLOV8 =====\")\nresults = metrics.results_dict\n\nprint(f\"Precision: {results['metrics/precision(B)']:.3f}\")\nprint(f\"Recall:    {results['metrics/recall(B)']:.3f}\")\nprint(f\"mAP@50:    {results['metrics/mAP50(B)']:.3f}\")\nprint(f\"mAP@50-95: {results['metrics/mAP50-95(B)']:.3f}\")\n\n# =========================\n# 2. SPECIFICITY TỪ CONFUSION MATRIX\n# =========================\n# metrics.confusion_matrix.matrix có shape (num_classes + 1, num_classes + 1)\n# Hàng = ground truth, Cột = prediction\n# Hàng/cột cuối cùng = background (no object)\n\ncm = metrics.confusion_matrix.matrix  # shape: (C+1, C+1)\nnum_classes = len(model.names)\n\nprint(\"\\n===== SPECIFICITY PER-CLASS =====\")\n\nspecificities = []\nrows = []\nnames = model.names\np = metrics.box.p\nr = metrics.box.r\nap50 = metrics.box.ap50\nap = metrics.box.ap\n\nfor i in range(num_classes):\n    # TP: cm[i, i]\n    # FN: tổng hàng i trừ TP (ground truth class i bị predict sai)\n    # FP: tổng cột i trừ TP (predict class i nhưng thực ra không phải)\n    # TN: tổng tất cả - TP - FP - FN\n\n    TP = cm[i, i]\n    FP = cm[:, i].sum() - TP      # cột i, bỏ TP\n    FN = cm[i, :].sum() - TP      # hàng i, bỏ TP\n    TN = cm.sum() - TP - FP - FN  # còn lại\n\n    specificity = TN / (TN + FP) if (TN + FP) > 0 else 0.0\n    specificities.append(specificity)\n\n    cls_name = names[i]\n    precision = float(p[i])\n    recall    = float(r[i])\n    ap_50     = float(ap50[i])\n    ap_5095   = float(ap[i])\n\n    print(f\"{cls_name:25s} | P: {precision:.3f} | R: {recall:.3f} \"\n          f\"| Spec: {specificity:.3f} | AP50: {ap_50:.3f} | AP50-95: {ap_5095:.3f}\")\n\n    rows.append({\n        \"class\":       cls_name,\n        \"precision\":   precision,\n        \"recall\":      recall,\n        \"specificity\": specificity,\n        \"AP50\":        ap_50,\n        \"AP50-95\":     ap_5095\n    })\n\n# =========================\n# 3. MACRO AVERAGE SPECIFICITY\n# =========================\nmacro_specificity = np.mean(specificities)\nprint(f\"\\n{'':25s}   Macro-avg Specificity: {macro_specificity:.3f}\")\n\n# =========================\n# 4. EXPORT DATAFRAME\n# =========================\ndf = pd.DataFrame(rows)\ndf.loc[len(df)] = {\n    \"class\":       \"MEAN\",\n    \"precision\":   df[\"precision\"].mean(),\n    \"recall\":      df[\"recall\"].mean(),\n    \"specificity\": macro_specificity,\n    \"AP50\":        df[\"AP50\"].mean(),\n    \"AP50-95\":     df[\"AP50-95\"].mean()\n}\n\nprint(\"\\n===== SUMMARY TABLE =====\")\nprint(df.to_string(index=False))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-17T03:50:44.943186Z","iopub.execute_input":"2026-05-17T03:50:44.943503Z","iopub.status.idle":"2026-05-17T03:55:52.792601Z","shell.execute_reply.started":"2026-05-17T03:50:44.943462Z","shell.execute_reply":"2026-05-17T03:55:52.791943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\nimport pandas as pd\n\n# Load model\nmodel = YOLO(\"/kaggle/input/version-16/best.pt\")\n\n# Run validation with high-recall settings\nmetrics = model.val(\n    data=\"vinbigdata.yaml\",\n    imgsz=1024,\n    split=\"val\",\n    conf=0.001,      # Hạ xuống mức tối thiểu để lấy trọn PR-curve\n    iou=0.65,        # Nới lỏng NMS, cho phép các hộp đè lên nhau nhiều hơn\n    augment=True,    # Kích hoạt TTA (Test-Time Augmentation)\n    max_det=300,     # (Tùy chọn) Đảm bảo không bị giới hạn số lượng hộp dự đoán\n    verbose=False\n)\n\n\n\n# =========================\n# 1. OVERALL METRICS\n# =========================\nprint(\"\\n===== OVERALL METRICS YOLOV8 =====\")\nresults = metrics.results_dict\n\nprint(f\"Precision: {results['metrics/precision(B)']:.3f}\")\nprint(f\"Recall:    {results['metrics/recall(B)']:.3f}\")\nprint(f\"mAP@50:    {results['metrics/mAP50(B)']:.3f}\")\nprint(f\"mAP@50-95: {results['metrics/mAP50-95(B)']:.3f}\")\n\n# =========================\n# 2. SPECIFICITY TỪ CONFUSION MATRIX\n# =========================\n# metrics.confusion_matrix.matrix có shape (num_classes + 1, num_classes + 1)\n# Hàng = ground truth, Cột = prediction\n# Hàng/cột cuối cùng = background (no object)\n\ncm = metrics.confusion_matrix.matrix  # shape: (C+1, C+1)\nnum_classes = len(model.names)\n\nprint(\"\\n===== SPECIFICITY PER-CLASS =====\")\n\nspecificities = []\nrows = []\nnames = model.names\np = metrics.box.p\nr = metrics.box.r\nap50 = metrics.box.ap50\nap = metrics.box.ap\n\nfor i in range(num_classes):\n    # TP: cm[i, i]\n    # FN: tổng hàng i trừ TP (ground truth class i bị predict sai)\n    # FP: tổng cột i trừ TP (predict class i nhưng thực ra không phải)\n    # TN: tổng tất cả - TP - FP - FN\n\n    TP = cm[i, i]\n    FP = cm[:, i].sum() - TP      # cột i, bỏ TP\n    FN = cm[i, :].sum() - TP      # hàng i, bỏ TP\n    TN = cm.sum() - TP - FP - FN  # còn lại\n\n    specificity = TN / (TN + FP) if (TN + FP) > 0 else 0.0\n    specificities.append(specificity)\n\n    cls_name = names[i]\n    precision = float(p[i])\n    recall    = float(r[i])\n    ap_50     = float(ap50[i])\n    ap_5095   = float(ap[i])\n\n    print(f\"{cls_name:25s} | P: {precision:.3f} | R: {recall:.3f} \"\n          f\"| Spec: {specificity:.3f} | AP50: {ap_50:.3f} | AP50-95: {ap_5095:.3f}\")\n\n    rows.append({\n        \"class\":       cls_name,\n        \"precision\":   precision,\n        \"recall\":      recall,\n        \"specificity\": specificity,\n        \"AP50\":        ap_50,\n        \"AP50-95\":     ap_5095\n    })\n\n# =========================\n# 3. MACRO AVERAGE SPECIFICITY\n# =========================\nmacro_specificity = np.mean(specificities)\nprint(f\"\\n{'':25s}   Macro-avg Specificity: {macro_specificity:.3f}\")\n\n# =========================\n# 4. EXPORT DATAFRAME\n# =========================\ndf = pd.DataFrame(rows)\ndf.loc[len(df)] = {\n    \"class\":       \"MEAN\",\n    \"precision\":   df[\"precision\"].mean(),\n    \"recall\":      df[\"recall\"].mean(),\n    \"specificity\": macro_specificity,\n    \"AP50\":        df[\"AP50\"].mean(),\n    \"AP50-95\":     df[\"AP50-95\"].mean()\n}\n\nprint(\"\\n===== SUMMARY TABLE =====\")\nprint(df.to_string(index=False))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-17T03:50:44.943186Z","iopub.execute_input":"2026-05-17T03:50:44.943503Z","iopub.status.idle":"2026-05-17T03:55:52.792601Z","shell.execute_reply.started":"2026-05-17T03:50:44.943462Z","shell.execute_reply":"2026-05-17T03:55:52.791943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from ultralytics import YOLO\n# import pandas as pd\n\n# # Load model\n# model = YOLO(\"/kaggle/input/datasets/ggduck14/train-yolov8-final/bestYolov8l_final.pt\")\n\n# # Run validation\n# metrics = model.val(\n#     data=\"vinbigdata.yaml\",\n#     imgsz=1024,\n#     split=\"val\",\n#     conf=0.20,\n#     iou=0.55,\n#     verbose=False\n# )\n\n# # =========================\n# # 1. OVERALL METRICS\n# # =========================\n# print(\"\\n===== OVERALL METRICS YOLOV8 =====\")\n# results = metrics.results_dict\n\n# print(f\"Precision: {results['metrics/precision(B)']:.3f}\")\n# print(f\"Recall:    {results['metrics/recall(B)']:.3f}\")\n# print(f\"mAP@50:    {results['metrics/mAP50(B)']:.3f}\")\n# print(f\"mAP@50-95: {results['metrics/mAP50-95(B)']:.3f}\")\n\n# # =========================\n# # 2. PER-CLASS FULL METRICS\n# # =========================\n# print(\"\\n===== PER-CLASS METRICS YOLOV8 =====\")\n\n# names = model.names\n# p = metrics.box.p\n# r = metrics.box.r\n# ap50 = metrics.box.ap50\n# ap = metrics.box.ap\n\n# rows = []\n\n# for i, cls_name in names.items():\n#     precision = float(p[i])\n#     recall = float(r[i])\n#     ap_50 = float(ap50[i])\n#     ap_5095 = float(ap[i])\n\n#     print(f\"{cls_name:25s} | P: {precision:.3f} | R: {recall:.3f} | AP50: {ap_50:.3f} | AP50-95: {ap_5095:.3f}\")\n\n#     rows.append({\n#         \"class\": cls_name,\n#         \"precision\": precision,\n#         \"recall\": recall,\n#         \"AP50\": ap_50,\n#         \"AP50-95\": ap_5095\n#     })","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-17T03:55:52.948934Z","iopub.status.idle":"2026-05-17T03:55:52.949902Z","shell.execute_reply.started":"2026-05-17T03:55:52.949681Z","shell.execute_reply":"2026-05-17T03:55:52.949717Z"}},"outputs":[],"execution_count":null}]}