{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":1809508,"sourceType":"datasetVersion","datasetId":1074034},{"sourceId":57614134,"sourceType":"kernelVersion"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-30T09:58:54.18939Z","iopub.execute_input":"2025-09-30T09:58:54.189612Z","iopub.status.idle":"2025-09-30T09:59:15.734491Z","shell.execute_reply.started":"2025-09-30T09:58:54.189593Z","shell.execute_reply":"2025-09-30T09:59:15.732309Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import kar rahe hain saare modules jo chahiye\nimport os                    # File system operations ke liye\nimport pydicom              # DICOM files read karne ke liye\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut  # Image quality ke liye\nimport cv2                  # Image processing (PNG save)\nimport numpy as np          # Array operations\nfrom PIL import Image       # Image visualization\nimport pandas as pd         # CSV handling\nimport matplotlib.pyplot as plt  # Plotting ke liye\nimport matplotlib.patches as patches  # Bounding boxes draw karne\nimport json                 # COCO JSON banane ke liye","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T09:59:47.308289Z","iopub.execute_input":"2025-09-30T09:59:47.308589Z","iopub.status.idle":"2025-09-30T09:59:47.314373Z","shell.execute_reply.started":"2025-09-30T09:59:47.308567Z","shell.execute_reply":"2025-09-30T09:59:47.313358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport cv2\nimport numpy as np\nfrom PIL import Image\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport json\nfrom multiprocessing import Pool  # Parallel processing ke liye","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dataset slug (sahi spelling)\nDATASET_SLUG = 'vinbigdata-chest-xray-abnormalities-detection'\n\n# Paths\nDICOM_DIR = f'/kaggle/input/{DATASET_SLUG}/train/'\nPNG_OUTPUT_DIR = '/kaggle/working/png_images/'\nANNOTATIONS_CSV = f'/kaggle/input/{DATASET_SLUG}/train.csv'\nYOLO_OUTPUT_DIR = '/kaggle/working/yolo_annotations/'\nCOCO_OUTPUT_JSON = '/kaggle/working/coco_annotations.json'\n\n# Output folders\nos.makedirs(PNG_OUTPUT_DIR, exist_ok=True)\nos.makedirs(YOLO_OUTPUT_DIR, exist_ok=True)\n\n# Paths verify\nprint(f\"Checking DICOM_DIR: {DICOM_DIR}\")\nprint(f\"DICOM_DIR exists: {os.path.exists(DICOM_DIR)}\")\nprint(f\"ANNOTATIONS_CSV exists: {os.path.exists(ANNOTATIONS_CSV)}\")\n\nif not (os.path.exists(DICOM_DIR) and os.path.exists(ANNOTATIONS_CSV)):\n    print(\"❌ ERROR: Dataset not added. Steps:\")\n    print(\"1. Click '+ Add Data' (right side).\")\n    print(\"2. Search 'vinbigdata-chest-xray-abnormalities-detection'.\")\n    print(\"3. Click 'Add'. Re-run this cell.\")\nelse:\n    dicom_count = len([f for f in os.listdir(DICOM_DIR) if f.endswith('.dcm')])\n    df = pd.read_csv(ANNOTATIONS_CSV)\n    print(f\"DICOM files: {dicom_count}\")\n    print(f\"CSV rows: {len(df)}, Columns: {list(df.columns)}\")\n    print(\"✅ Paths OK! Run next cells.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}