{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":12888836,"datasetId":8069325,"databundleVersionId":13537243},{"sourceType":"datasetVersion","sourceId":12758783,"datasetId":8065124,"databundleVersionId":13380061},{"sourceType":"datasetVersion","sourceId":12889909,"datasetId":8154439,"databundleVersionId":13538492},{"sourceType":"datasetVersion","sourceId":12853590,"datasetId":7999273,"databundleVersionId":13491956},{"sourceType":"kernelVersion","sourceId":255851339}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\n# paths to your two folders\nfolder1 = \"/kaggle/input/modified-ourpacs-5cls-360/train/train/images\"\nfolder2 = \"/kaggle/input/our-data-1000/train/images_old_total\"\n\n# get set of filenames (without paths)\nfiles1 = set(os.listdir(folder1))\nfiles2 = set(os.listdir(folder2))\n\n# find intersection (matched filenames)\nmatched = files1.intersection(files2)\n\nprint(f\"Number of matched images: {len(matched)}\")\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T17:55:53.823991Z","iopub.execute_input":"2025-08-27T17:55:53.824294Z","iopub.status.idle":"2025-08-27T17:55:53.83176Z","shell.execute_reply.started":"2025-08-27T17:55:53.824266Z","shell.execute_reply":"2025-08-27T17:55:53.831069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nfrom tqdm import tqdm\n\n# paths to your two folders\nfolder1 = \"/kaggle/input/our-data-1000/train/images\"   # correct names\nfolder2 = \"/kaggle/input/modified-ourpacs-5cls-360/train/train/images\"  # long names\noutput_folder = \"/kaggle/working/exceed_folder1_output\"\n\n# create output folder if not exist\nos.makedirs(output_folder, exist_ok=True)\n\n# get filenames\nfiles1 = os.listdir(folder1)\nfiles2 = os.listdir(folder2)\n\n# remove extensions for folder1\nids1 = {os.path.splitext(f)[0]: f for f in files1}\n\n# extract first number (before _ or -) from folder2\ndef get_first_number(filename):\n    return filename.split(\"_\")[0].split(\"-\")[0]\n\nids2 = {get_first_number(f): f for f in files2}\n\n# find matches\nmatched_keys = set(ids1.keys()).intersection(ids2.keys())\n\n# find exceed/missing\nexceed_in_folder1 = set(ids1.keys()) - set(ids2.keys())  # only in folder1\nexceed_in_folder2 = set(ids2.keys()) - set(ids1.keys())  # only in folder2\n\nprint(f\"✅ Number of matched images: {len(matched_keys)}\")\nprint(f\"📂 Exceed in folder1 (only in folder1): {len(exceed_in_folder1)}\")\nprint(f\"📂 Exceed in folder2 (only in folder2): {len(exceed_in_folder2)}\")\n\n# copy exceed images from folder1 to output folder with progress bar\nprint(\"\\n📥 Copying exceed images from folder1...\")\nfor key in tqdm(exceed_in_folder1, desc=\"Copying\"):\n    img_name = ids1[key]\n    src = os.path.join(folder1, img_name)\n    dst = os.path.join(output_folder, img_name)\n    if os.path.isfile(src):\n        shutil.copy(src, dst)\n\nprint(f\"\\n✅ Done! Exceed images from folder1 are saved in: {output_folder}\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-27T17:57:57.612952Z","iopub.execute_input":"2025-08-27T17:57:57.61322Z","iopub.status.idle":"2025-08-27T17:58:05.456776Z","shell.execute_reply.started":"2025-08-27T17:57:57.613198Z","shell.execute_reply":"2025-08-27T17:58:05.455824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ultralytics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T18:18:00.136912Z","iopub.execute_input":"2025-08-27T18:18:00.137374Z","iopub.status.idle":"2025-08-27T18:18:03.709489Z","shell.execute_reply.started":"2025-08-27T18:18:00.137349Z","shell.execute_reply":"2025-08-27T18:18:03.708648Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\nimport torch\nimport numpy as np\nfrom ultralytics import YOLO\nfrom tqdm import tqdm\n\n# --------------------------\n# CONFIG\n# --------------------------\nimages_folder = \"/kaggle/working/exceed_folder1_output\"   # folder of test images\nmodel_path = \"/kaggle/input/finetuned-models/best_5epoch.pt\"\n\n# Load YOLO model\nmodel = YOLO(model_path)\n\n# Thresholds per class\nclass_thresholds = {\n    0: 0.360,   # Aortic Enlargement\n    1: 0.180,   # Cardiomegaly\n    2: 0.000,   # Lung Opacity\n    3: 0.090,   # Pleural Effusion\n    4: 0.10    # Fifth class (if exists)\n}\n\nvalid_classes = set(class_thresholds.keys())\n\n# --------------------------\n# COUNTS\n# --------------------------\npredicted_counts = {cls: 0 for cls in valid_classes}\n\nimage_files = glob.glob(os.path.join(images_folder, \"*.jpg\"))  # change to *.png if needed\n\nfor img_path in tqdm(image_files, desc=\"Predicting\"):\n    results = model(img_path, verbose=False)[0]\n\n    pred_classes = set()\n    for box in results.boxes:\n        cls = int(box.cls.cpu().numpy().item())\n        conf = float(box.conf.cpu().numpy().item())\n\n        if cls in valid_classes:\n            if conf >= class_thresholds[cls]:\n                pred_classes.add(cls)\n\n    # add counts: one image can count to multiple classes\n    for cls in pred_classes:\n        predicted_counts[cls] += 1\n\n# --------------------------\n# PRINT RESULT\n# --------------------------\nprint(\"\\n📊 Predicted Images Per Class:\")\nfor cls in valid_classes:\n    print(f\"Class {cls}: {predicted_counts[cls]} images\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T18:27:30.809127Z","iopub.execute_input":"2025-08-27T18:27:30.809495Z","iopub.status.idle":"2025-08-27T18:28:13.869376Z","shell.execute_reply.started":"2025-08-27T18:27:30.809468Z","shell.execute_reply":"2025-08-27T18:28:13.868545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom tqdm import tqdm\nfrom collections import defaultdict\n\n# Paths\nimages_dir = \"/kaggle/working/exceed_folder1_output\"   # change to your images folder\nlabels_dir = \"/kaggle/input/our-data-1000/train/labels\"   # change to your labels folder\n\n# Collect images that have a matching .txt\nimage_extensions = [\".jpg\", \".jpeg\", \".png\"]\nmatched_images = []\n\nfor img_name in tqdm(os.listdir(images_dir), desc=\"Matching images\"):\n    base, ext = os.path.splitext(img_name)\n    if ext.lower() in image_extensions:\n        label_path = os.path.join(labels_dir, base + \".txt\")\n        if os.path.exists(label_path):  # keep only images with labels\n            matched_images.append((img_name, label_path))\n\nprint(f\"\\n✅ Total matched images with labels: {len(matched_images)}\")\n\n# Count per class\nclass_counts = defaultdict(int)\n\nfor img_name, label_path in tqdm(matched_images, desc=\"Counting classes\"):\n    with open(label_path, \"r\") as f:\n        for line in f:\n            parts = line.strip().split()\n            if not parts:\n                continue\n            cls = parts[0]\n            # drop class \"4\" if it appears\n            if cls != \"4\":\n                class_counts[cls] += 1\n\nprint(\"\\n📊 Number of images per class:\")\nfor cls, count in sorted(class_counts.items(), key=lambda x: int(x[0])):\n    print(f\"Class {cls}: {count}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T18:30:58.935748Z","iopub.execute_input":"2025-08-27T18:30:58.936391Z","iopub.status.idle":"2025-08-27T18:31:02.538141Z","shell.execute_reply.started":"2025-08-27T18:30:58.936364Z","shell.execute_reply":"2025-08-27T18:31:02.537285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\nimport numpy as np\nfrom ultralytics import YOLO\nfrom sklearn.metrics import precision_score, recall_score, f1_score\nfrom tqdm import tqdm\n\n# --------------------------\n# CONFIG\n# --------------------------\nimages_folder = \"/kaggle/working/exceed_folder1_output\"   # folder of test images\nlabels_folder = \"/kaggle/input/our-data-1000/train/labels\"   # folder of txt labels\nmodel_path = \"/kaggle/input/finetuned-models/best_5epoch.pt\"\n\n# Load YOLO model\nmodel = YOLO(model_path)\n\n# Thresholds per class\nclass_thresholds = {\n    0: 0.360,   # Aortic Enlargement\n    1: 0.180,   # Cardiomegaly\n    2: 0.000,   # Lung Opacity\n    3: 0.090,   # Pleural Effusion\n    4: 0.100    # Fifth class (if exists)\n}\n\nvalid_classes = set(class_thresholds.keys())\n\n# --------------------------\n# CLASS NAMES\n# --------------------------\nclass_names = {\n    0: \"Aortic Enlargement\",\n    1: \"Cardiomegaly\",\n    2: \"Lung Opacity\",\n    3: \"Pleural Effusion\",\n    4: \"Other\"\n}\n\n# --------------------------\n# MATCHED IMAGES\n# --------------------------\nimage_extensions = [\".jpg\", \".jpeg\", \".png\"]\nmatched_images = []\n\nfor img_name in os.listdir(images_folder):\n    base, ext = os.path.splitext(img_name)\n    if ext.lower() in image_extensions:\n        label_path = os.path.join(labels_folder, base + \".txt\")\n        if os.path.exists(label_path):\n            matched_images.append((os.path.join(images_folder, img_name), label_path))\n\nprint(f\"\\n✅ Total matched images with labels: {len(matched_images)}\")\n\n# --------------------------\n# COLLECT Y_TRUE / Y_PRED\n# --------------------------\ny_true = []\ny_pred = []\n\nfor img_path, label_path in tqdm(matched_images, desc=\"Evaluating\"):\n    # --- Ground truth ---\n    true_classes = set()\n    with open(label_path, \"r\") as f:\n        for line in f:\n            parts = line.strip().split()\n            if not parts:\n                continue\n            cls = int(parts[0])\n            if cls in valid_classes:\n                true_classes.add(cls)\n\n    # --- Predictions ---\n    results = model(img_path, verbose=False)[0]\n\n    pred_classes = set()\n    for box in results.boxes:\n        cls = int(box.cls.cpu().numpy().item())\n        conf = float(box.conf.cpu().numpy().item())\n        if cls in valid_classes and conf >= class_thresholds[cls]:\n            pred_classes.add(cls)\n\n    # one-hot encoding (image-level)\n    gt_vec = [1 if cls in true_classes else 0 for cls in valid_classes]\n    pr_vec = [1 if cls in pred_classes else 0 for cls in valid_classes]\n\n    y_true.append(gt_vec)\n    y_pred.append(pr_vec)\n\n# convert to numpy\ny_true = np.array(y_true)\ny_pred = np.array(y_pred)\n\n# --------------------------\n# METRICS\n# --------------------------\nprint(\"\\n📊 Per-class metrics:\")\nfor i, cls in enumerate(sorted(valid_classes)):\n    cls_true = y_true[:, i]\n    cls_pred = y_pred[:, i]\n\n    p = precision_score(cls_true, cls_pred, zero_division=0)\n    r = recall_score(cls_true, cls_pred, zero_division=0)\n    f1 = f1_score(cls_true, cls_pred, zero_division=0)\n\n    print(f\"{class_names[cls]}: Precision={p:.3f}, Recall={r:.3f}, F1={f1:.3f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T18:45:51.279997Z","iopub.execute_input":"2025-08-27T18:45:51.280741Z","iopub.status.idle":"2025-08-27T18:46:37.541597Z","shell.execute_reply.started":"2025-08-27T18:45:51.280717Z","shell.execute_reply":"2025-08-27T18:46:37.540898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nfrom ultralytics import YOLO\nfrom sklearn.metrics import precision_score, recall_score, f1_score\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\n# --------------------------\n# CONFIG\n# --------------------------\nimages_folder = \"/kaggle/working/exceed_folder1_output\"   # test images\nlabels_folder = \"/kaggle/input/our-data-1000/train/labels\"   # ground-truth labels\nmodel_path = \"/kaggle/input/finetuned-models/best_5epoch.pt\"\n\n# Load YOLO model\nmodel = YOLO(model_path)\n\n# Classes (0–4)\nvalid_classes = [0, 1, 2, 3, 4]\n\n# --------------------------\n# MATCHED IMAGES\n# --------------------------\nimage_extensions = [\".jpg\", \".jpeg\", \".png\"]\nmatched_images = []\nfor img_name in os.listdir(images_folder):\n    base, ext = os.path.splitext(img_name)\n    if ext.lower() in image_extensions:\n        label_path = os.path.join(labels_folder, base + \".txt\")\n        if os.path.exists(label_path):\n            matched_images.append((os.path.join(images_folder, img_name), label_path))\n\nprint(f\"\\n✅ Total matched images with labels: {len(matched_images)}\")\n\n# --------------------------\n# STORE ALL PREDICTIONS & GT\n# --------------------------\nall_true = []\nall_pred_conf = []   # هنا نخزن confidences مش بس 0/1\n\nfor img_path, label_path in tqdm(matched_images, desc=\"Running model\"):\n    # --- Ground truth ---\n    true_classes = set()\n    with open(label_path, \"r\") as f:\n        for line in f:\n            parts = line.strip().split()\n            if parts:\n                cls = int(parts[0])\n                if cls in valid_classes:\n                    true_classes.add(cls)\n\n    # --- Predictions ---\n    results = model(img_path, verbose=False)[0]\n    pred_conf = {cls: 0.0 for cls in valid_classes}  # maximum conf per class\n\n    for box in results.boxes:\n        cls = int(box.cls.cpu().numpy().item())\n        conf = float(box.conf.cpu().numpy().item())\n        if cls in valid_classes:\n            pred_conf[cls] = max(pred_conf[cls], conf)  # ناخد أعلى conf لو أكتر من واحد\n\n    # one-hot GT\n    gt_vec = [1 if cls in true_classes else 0 for cls in valid_classes]\n    all_true.append(gt_vec)\n    all_pred_conf.append([pred_conf[cls] for cls in valid_classes])\n\nall_true = np.array(all_true)\nall_pred_conf = np.array(all_pred_conf)\n\n# --------------------------\n# SWEEP THRESHOLDS\n# --------------------------\nthresholds = np.arange(0.0, 1.01, 0.05)  # من 0 لـ 1 بخطوة 0.05\n\nmetrics = {cls: {\"precision\": [], \"recall\": [], \"f1\": []} for cls in valid_classes}\n\nfor t in thresholds:\n    for i, cls in enumerate(valid_classes):\n        cls_true = all_true[:, i]\n        cls_pred = (all_pred_conf[:, i] >= t).astype(int)\n\n        p = precision_score(cls_true, cls_pred, zero_division=0)\n        r = recall_score(cls_true, cls_pred, zero_division=0)\n        f1 = f1_score(cls_true, cls_pred, zero_division=0)\n\n        metrics[cls][\"precision\"].append(p)\n        metrics[cls][\"recall\"].append(r)\n        metrics[cls][\"f1\"].append(f1)\n\n# --------------------------\n# --------------------------\n# CLASS NAMES\n# --------------------------\nclass_names = {\n    0: \"Aortic Enlargement\",\n    1: \"Cardiomegaly\",\n    2: \"Lung Opacity\",\n    3: \"Pleural Effusion\",\n    4: \"Other\"   # لو في كلاس خامس\n}\n# --------------------------\n# --------------------------\n# PLOT\n# --------------------------\nfor cls in valid_classes:\n    plt.figure(figsize=(8,5))\n    plt.plot(thresholds, metrics[cls][\"precision\"], label=\"Precision\")\n    plt.plot(thresholds, metrics[cls][\"recall\"], label=\"Recall\")\n    plt.plot(thresholds, metrics[cls][\"f1\"], label=\"F1\")\n    plt.xlabel(\"Threshold\")\n    plt.ylabel(\"Score\")\n    plt.title(f\"{class_names[cls]} (Class {cls}) - Metrics vs Threshold\")\n    plt.legend()\n    plt.grid(True)\n    plt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T18:41:19.427862Z","iopub.execute_input":"2025-08-27T18:41:19.428161Z","iopub.status.idle":"2025-08-27T18:42:06.92414Z","shell.execute_reply.started":"2025-08-27T18:41:19.42814Z","shell.execute_reply":"2025-08-27T18:42:06.923183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ultralytics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T19:32:00.617468Z","iopub.execute_input":"2025-08-27T19:32:00.61774Z","iopub.status.idle":"2025-08-27T19:33:17.613762Z","shell.execute_reply.started":"2025-08-27T19:32:00.61772Z","shell.execute_reply":"2025-08-27T19:33:17.612991Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Train yolo Model**","metadata":{}},{"cell_type":"code","source":"# Paths to your split folders\ntrain_images = \"/kaggle/input/our-data-1000/data/data/images/train\"\nval_images   = \"/kaggle/input/our-data-1000/data/data/images/val\"\ntest_images  = \"/kaggle/input/our-data-1000/data/data/images/test\"\n\ntrain_labels = \"/kaggle/input/our-data-1000/data/data/labels/train\"\nval_labels   = \"/kaggle/input/our-data-1000/data/data/labels/val\"\ntest_labels  = \"/kaggle/input/our-data-1000/data/data/labels/test\"\n\ndata_yaml = \"data.yaml\"\n\nwith open(data_yaml, \"w\") as f:\n    f.write(f\"train: {train_images}\\n\")\n    f.write(f\"val: {val_images}\\n\")\n    f.write(f\"test: {test_images}\\n\")  # اختياري، بعض النسخ من YOLO تدعم test\n    f.write(\"nc: 3\\n\")\n    f.write(\"names: ['Aortic Enlargement', 'Cardiomegaly', 'Pleural Effusion']\\n\")\n\nprint(f\"✅ YAML file saved: {data_yaml}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T20:27:46.905729Z","iopub.execute_input":"2025-08-27T20:27:46.906816Z","iopub.status.idle":"2025-08-27T20:27:46.914161Z","shell.execute_reply.started":"2025-08-27T20:27:46.906782Z","shell.execute_reply":"2025-08-27T20:27:46.913053Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --------------------------\n# لو عايز تبدأ من pre-trained model\nmodel = YOLO(\"/kaggle/input/yolooo/yolo11m.pt\")  # صغير للتجربة، أو استخدم موديل أكبر yolov8s.pt\n\n# --------------------------\n# TRAINING\n# --------------------------\nmodel.train(\n    data=data_yaml,      # ملف البيانات\n    epochs=50,           # عدد ال epochs\n    imgsz=640,           # حجم الصورة\n    batch=16,            # حجم الباتش\n    device=0,            # GPU:0, CPU: -1\n    name=\"my_yolo_model\" # اسم المشروع\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T20:27:52.474417Z","iopub.execute_input":"2025-08-27T20:27:52.474747Z","iopub.status.idle":"2025-08-27T20:54:25.368713Z","shell.execute_reply.started":"2025-08-27T20:27:52.474723Z","shell.execute_reply":"2025-08-27T20:54:25.367535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_best = YOLO(\"/kaggle/working/runs/detect/my_yolo_model2/weights/best.pt\")\n\n# --------------------------\n# EVALUATE on validation set\n# --------------------------\nval_results = model_best.val(data=data_yaml, split=\"val\", verbose=True)\nprint(\"\\n✅ Validation Results:\")\nprint(val_results)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T21:21:14.649916Z","iopub.execute_input":"2025-08-27T21:21:14.650311Z","iopub.status.idle":"2025-08-27T21:21:24.110176Z","shell.execute_reply.started":"2025-08-27T21:21:14.650277Z","shell.execute_reply":"2025-08-27T21:21:24.109272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# EVALUATE on test set\n# --------------------------\ntest_results = model_best.val(data=data_yaml, split=\"test\", verbose=True)\nprint(\"\\n✅ Test Results:\")\nprint(test_results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T21:21:53.594584Z","iopub.execute_input":"2025-08-27T21:21:53.594964Z","iopub.status.idle":"2025-08-27T21:22:04.281083Z","shell.execute_reply.started":"2025-08-27T21:21:53.594931Z","shell.execute_reply":"2025-08-27T21:22:04.280084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nfrom ultralytics import YOLO\nfrom sklearn.metrics import precision_score, recall_score, f1_score\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\n# --------------------------\n# CONFIG\n# --------------------------\ntrain_images_folder = \"/kaggle/input/our-data-1000/data/data/images/test\"   # train images\ntrain_labels_folder = \"/kaggle/input/our-data-1000/data/data/labels/test\"   # true labels\nmodel_path = \"/kaggle/working/runs/detect/my_yolo_model2/weights/best.pt\"           # trained YOLO model\n\n# Load YOLO model\nmodel = YOLO(model_path)\n\n# Classes\nclass_names = [\"Aortic Enlargement\", \"Cardiomegaly\", \"Pleural Effusion\"]\nvalid_classes = list(range(len(class_names)))  # [0,1,2]\n\n# --------------------------\n# MATCHED IMAGES\n# --------------------------\nimage_extensions = [\".jpg\", \".jpeg\", \".png\"]\nmatched_images = []\n\nfor img_name in os.listdir(train_images_folder):\n    base, ext = os.path.splitext(img_name)\n    if ext.lower() in image_extensions:\n        label_path = os.path.join(train_labels_folder, base + \".txt\")\n        if os.path.exists(label_path):\n            matched_images.append((os.path.join(train_images_folder, img_name), label_path))\n\nprint(f\"✅ Total matched train images: {len(matched_images)}\")\n\n# --------------------------\n# COLLECT ALL CONFIDENCES AND GT\n# --------------------------\nall_true = []\nall_pred_conf = []\n\nfor img_path, label_path in tqdm(matched_images, desc=\"Running model on train\"):\n    # Ground truth\n    true_classes = set()\n    with open(label_path, \"r\") as f:\n        for line in f:\n            parts = line.strip().split()\n            if parts:\n                cls = int(parts[0])\n                if cls in valid_classes:\n                    true_classes.add(cls)\n    gt_vec = [1 if cls in true_classes else 0 for cls in valid_classes]\n    all_true.append(gt_vec)\n\n    # Model predictions\n    results = model(img_path, verbose=False)[0]\n    pred_conf = {cls: 0.0 for cls in valid_classes}\n    for box in results.boxes:\n        cls = int(box.cls.cpu().numpy().item())\n        conf = float(box.conf.cpu().numpy().item())\n        if cls in valid_classes:\n            pred_conf[cls] = max(pred_conf[cls], conf)  # highest confidence per class\n    all_pred_conf.append([pred_conf[cls] for cls in valid_classes])\n\nall_true = np.array(all_true)\nall_pred_conf = np.array(all_pred_conf)\n\n# --------------------------\n# SWEEP THRESHOLDS WITH PROGRESS BAR\n# --------------------------\nthresholds = np.arange(0.0, 1.01, 0.05)\nmetrics = {cls: {\"precision\": [], \"recall\": [], \"f1\": []} for cls in valid_classes}\n\nfor t in tqdm(thresholds, desc=\"Sweeping thresholds\"):\n    for i, cls in enumerate(valid_classes):\n        cls_true = all_true[:, i]\n        cls_pred = (all_pred_conf[:, i] >= t).astype(int)\n        p = precision_score(cls_true, cls_pred, zero_division=0)\n        r = recall_score(cls_true, cls_pred, zero_division=0)\n        f1 = f1_score(cls_true, cls_pred, zero_division=0)\n        metrics[cls][\"precision\"].append(p)\n        metrics[cls][\"recall\"].append(r)\n        metrics[cls][\"f1\"].append(f1)\n\n# --------------------------\n# PLOT METRICS PER CLASS\n# --------------------------\nfor cls in valid_classes:\n    plt.figure(figsize=(8,5))\n    plt.plot(thresholds, metrics[cls][\"precision\"], label=\"Precision\")\n    plt.plot(thresholds, metrics[cls][\"recall\"], label=\"Recall\")\n    plt.plot(thresholds, metrics[cls][\"f1\"], label=\"F1\")\n    plt.xlabel(\"Threshold\")\n    plt.ylabel(\"Score\")\n    plt.title(f\"{class_names[cls]} - Metrics vs Threshold\")\n    plt.legend()\n    plt.grid(True)\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T21:40:37.720195Z","iopub.execute_input":"2025-08-27T21:40:37.720899Z","iopub.status.idle":"2025-08-27T21:40:46.603646Z","shell.execute_reply.started":"2025-08-27T21:40:37.720875Z","shell.execute_reply":"2025-08-27T21:40:46.602617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nfrom ultralytics import YOLO\nfrom sklearn.metrics import precision_score, recall_score, f1_score\nfrom tqdm import tqdm\n\n# --------------------------\n# CONFIG\n# --------------------------\nval_images = \"/kaggle/input/our-data-1000/data/data/images/test\"  \nval_labels = \"/kaggle/input/our-data-1000/data/data/labels/test\"\nmodel_path = \"/kaggle/working/runs/detect/my_yolo_model2/weights/best.pt\"\n\n# Load model\nmodel = YOLO(model_path)\n\n# Classes & best thresholds (from training)\nvalid_classes = [0, 1, 2]  # بعد إزالة Lung Opacity\nbest_thresholds = {\n    0: 0.42,\n    1: 0.39,\n    2: 0.09\n}\n\n# --------------------------\n# MATCHED IMAGES\n# --------------------------\nimage_extensions = [\".jpg\", \".jpeg\", \".png\"]\nmatched_images = []\n\nfor img_name in os.listdir(val_images):\n    base, ext = os.path.splitext(img_name)\n    if ext.lower() in image_extensions:\n        label_path = os.path.join(val_labels, base + \".txt\")\n        if os.path.exists(label_path):\n            matched_images.append((os.path.join(val_images, img_name), label_path))\n\nprint(f\"✅ Total matched validation images: {len(matched_images)}\")\n\n# --------------------------\n# COLLECT Y_TRUE / Y_PRED\n# --------------------------\ny_true = []\ny_pred = []\n\nfor img_path, label_path in tqdm(matched_images, desc=\"Evaluating on val\"):\n    # Ground truth\n    true_classes = set()\n    with open(label_path, \"r\") as f:\n        for line in f:\n            parts = line.strip().split()\n            if parts:\n                cls = int(parts[0])\n                if cls in valid_classes:\n                    true_classes.add(cls)\n\n    # Predictions\n    results = model(img_path, verbose=False)[0]\n    pred_classes = set()\n    for box in results.boxes:\n        cls = int(box.cls.cpu().numpy().item())\n        conf = float(box.conf.cpu().numpy().item())\n        if cls in valid_classes and conf >= best_thresholds[cls]:\n            pred_classes.add(cls)\n\n    # One-hot encoding\n    gt_vec = [1 if cls in true_classes else 0 for cls in valid_classes]\n    pr_vec = [1 if cls in pred_classes else 0 for cls in valid_classes]\n\n    y_true.append(gt_vec)\n    y_pred.append(pr_vec)\n\ny_true = np.array(y_true)\ny_pred = np.array(y_pred)\n\n# --------------------------\n# METRICS\n# --------------------------\n# --------------------------\n# METRICS\n# --------------------------\nclass_names = {\n    0: \"Aortic Enlargement\",\n    1: \"Cardiomegaly\",\n    2: \"Pleural Effusion\"\n}\n\nprint(\"\\n📊 Validation set metrics (using best thresholds):\")\nfor i, cls in enumerate(valid_classes):\n    cls_true = y_true[:, i]\n    cls_pred = y_pred[:, i]\n\n    p = precision_score(cls_true, cls_pred, zero_division=0)\n    r = recall_score(cls_true, cls_pred, zero_division=0)\n    f1 = f1_score(cls_true, cls_pred, zero_division=0)\n\n    print(f\"{class_names[cls]} (Class {cls}): Precision={p:.3f}, Recall={r:.3f}, F1={f1:.3f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T21:39:37.804186Z","iopub.execute_input":"2025-08-27T21:39:37.804694Z","iopub.status.idle":"2025-08-27T21:39:45.908546Z","shell.execute_reply.started":"2025-08-27T21:39:37.804663Z","shell.execute_reply":"2025-08-27T21:39:45.907528Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Loading Vinbig Data and our Data**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Load CSV\ncsv_path = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv\"\ndf = pd.read_csv(csv_path)\n\n# لو كل صورة ممكن يكون لها أكثر من class (multi-label) خليه unique على image + class\ndf_unique = df[['image_id', 'class_id', 'class_name']].drop_duplicates()\n\n# Count number of unique images per class\nimage_counts = df_unique.groupby(['class_id', 'class_name'])['image_id'].nunique()\n\nprint(\"📊 Number of images per class:\")\nprint(image_counts)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T12:24:32.573683Z","iopub.execute_input":"2025-08-28T12:24:32.573932Z","iopub.status.idle":"2025-08-28T12:24:32.77236Z","shell.execute_reply.started":"2025-08-28T12:24:32.573914Z","shell.execute_reply":"2025-08-28T12:24:32.771628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load CSV\n# csv_path = \"path/to/vinbigdata_annotations.csv\"\n# df = pd.read_csv(csv_path)\n\n# Classes to keep\nkeep_classes = ['Aortic enlargement', 'Pleural effusion']\n\n# Filter the dataframe\ndf_filtered = df[df['class_name'].isin(keep_classes)].copy()\n\n# Optionally, drop duplicates if an image has the same class multiple times\ndf_filtered = df_filtered[['image_id', 'class_id', 'class_name']].drop_duplicates()\n\n# Count number of images per class after filtering\nimage_counts = df_filtered.groupby(['class_id', 'class_name'])['image_id'].nunique()\nprint(\"📊 Number of images per class (after filtering):\")\nprint(image_counts)\n\n# Optionally, save the filtered CSV\ndf_filtered.to_csv(\"vinbigdata_filtered.csv\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T12:24:35.261118Z","iopub.execute_input":"2025-08-28T12:24:35.261379Z","iopub.status.idle":"2025-08-28T12:24:35.291847Z","shell.execute_reply.started":"2025-08-28T12:24:35.261358Z","shell.execute_reply":"2025-08-28T12:24:35.291272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport pydicom\nimport cv2\nimport numpy as np\nfrom tqdm import tqdm  # <-- import tqdm\n\n# --------------------------\n# CONFIG\n# --------------------------\ncsv_path = \"vinbigdata_filtered.csv\"\ndicom_folder = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train\"\noutput_folder = \"/kaggle/working/filtered_images_png\"\nos.makedirs(output_folder, exist_ok=True)\n\ndf = pd.read_csv(csv_path)\nimage_ids = df['image_id'].unique()\n\ndef dicom_to_png_auto_invert(dicom_path):\n    ds = pydicom.dcmread(dicom_path)\n    img = ds.pixel_array.astype(float)\n\n    # Normalize to 0-255\n    img = (img - np.min(img)) / (np.max(img) - np.min(img))\n    img = (img * 255).astype(np.uint8)\n\n    # Auto invert contrast if image looks inverted\n    h, w = img.shape\n    center_region = img[h//4:3*h//4, w//4:3*w//4]\n    mean_center = np.mean(center_region)\n\n    if mean_center > 127:\n        img = 255 - img\n\n    return img\n\n# --------------------------\n# Convert all images with progress bar\n# --------------------------\nfor img_id in tqdm(image_ids, desc=\"Converting DICOMs to PNG\"):\n    dicom_path = os.path.join(dicom_folder, img_id + \".dicom\")\n    if not os.path.exists(dicom_path):\n        print(f\"❌ Missing DICOM: {dicom_path}\")\n        continue\n\n    img = dicom_to_png_auto_invert(dicom_path)\n    output_path = os.path.join(output_folder, img_id + \".png\")\n    cv2.imwrite(output_path, img)\n\nprint(f\"✅ Finished converting {len(image_ids)} images to PNG with auto-invert.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T12:24:37.380012Z","iopub.execute_input":"2025-08-28T12:24:37.380271Z","iopub.status.idle":"2025-08-28T13:31:33.460958Z","shell.execute_reply.started":"2025-08-28T12:24:37.380252Z","shell.execute_reply":"2025-08-28T13:31:33.4602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport zipfile\nfrom tqdm import tqdm\n\n# --------------------------\n# CONFIG\n# --------------------------\nfolder_to_zip = \"/kaggle/working/filtered_images_png\"  # folder containing your images\noutput_zip = \"/kaggle/working/images_archive.zip\"      # output zip file\n\n# Gather all files first\nall_files = []\nfor root, dirs, files in os.walk(folder_to_zip):\n    for file in files:\n        all_files.append(os.path.join(root, file))\n\n# Create a Zip file and add files one by one\nwith zipfile.ZipFile(output_zip, 'w', zipfile.ZIP_DEFLATED) as zipf:\n    for file_path in tqdm(all_files, desc=\"Zipping images\"):\n        zipf.write(file_path, os.path.relpath(file_path, folder_to_zip))\n        os.remove(file_path)  # delete original file to save space\n\nprint(f\"✅ Folder zipped successfully! Saved at: {output_zip}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T13:31:53.427813Z","iopub.execute_input":"2025-08-28T13:31:53.428089Z","iopub.status.idle":"2025-08-28T13:39:00.583356Z","shell.execute_reply.started":"2025-08-28T13:31:53.428067Z","shell.execute_reply":"2025-08-28T13:39:00.582711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kaggle kernels output ahmedahmoud/tst-our-data-1000-using-finetuned-5epoch -p /path/to/dest","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}