{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":99552,"databundleVersionId":13851420,"sourceType":"competition"},{"sourceId":1421668,"sourceType":"datasetVersion","datasetId":832340}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\nDEBUG = True  \nGLOBAL_WIDTH = 224 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:26.522833Z","iopub.execute_input":"2025-10-13T01:04:26.523104Z","iopub.status.idle":"2025-10-13T01:04:26.52673Z","shell.execute_reply.started":"2025-10-13T01:04:26.523085Z","shell.execute_reply":"2025-10-13T01:04:26.526052Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Install and import Libraries","metadata":{}},{"cell_type":"code","source":"!pip install dicomsdl\n!pip install python-gdcm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:26.527714Z","iopub.execute_input":"2025-10-13T01:04:26.527991Z","iopub.status.idle":"2025-10-13T01:04:30.379422Z","shell.execute_reply.started":"2025-10-13T01:04:26.527974Z","shell.execute_reply":"2025-10-13T01:04:30.378647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport re\nimport glob\nimport time\n\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import convert_color_space\nimport dicomsdl as dicoml\nfrom matplotlib import pyplot as plt\nfrom joblib import Parallel, delayed\nfrom IPython.display import HTML\nfrom multiprocessing import Pool, cpu_count\nimport imageio\nimport shutil","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:33.390876Z","iopub.execute_input":"2025-10-13T01:04:33.391137Z","iopub.status.idle":"2025-10-13T01:04:33.396737Z","shell.execute_reply.started":"2025-10-13T01:04:33.391104Z","shell.execute_reply":"2025-10-13T01:04:33.396006Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Init some variables and helper functions","metadata":{}},{"cell_type":"code","source":"DATA_ROOT = '/kaggle/input/rsna-intracranial-aneurysm-detection'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:33.412696Z","iopub.execute_input":"2025-10-13T01:04:33.412926Z","iopub.status.idle":"2025-10-13T01:04:33.426142Z","shell.execute_reply.started":"2025-10-13T01:04:33.41291Z","shell.execute_reply":"2025-10-13T01:04:33.425405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ds = pydicom.dcmread(path, stop_before_pixels=True)\n\n#         # Try to use InstanceNumber first\n# image_position = getattr(ds, 'ImagePositionPatient', None)\n# print(image_position)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:33.426843Z","iopub.execute_input":"2025-10-13T01:04:33.427121Z","iopub.status.idle":"2025-10-13T01:04:33.439162Z","shell.execute_reply.started":"2025-10-13T01:04:33.427104Z","shell.execute_reply":"2025-10-13T01:04:33.438636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from joblib import Parallel, delayed, cpu_count\n\n# # Note: assuming extract_sort_key is defined elsewhere\n# def fast_sort_dicom_paths_joblib(dcm_paths, n_jobs=None):\n#     if n_jobs is None:\n#         n_jobs = max(1, cpu_count() - 1)\n\n#     # Use Parallel and delayed to execute the function on all cores\n#     sort_info = Parallel(n_jobs=n_jobs)(\n#         delayed(extract_sort_key)(path) for path in dcm_paths\n#     )\n\n#     sort_info.sort()\n#     return [x[2] for x in sort_info]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:33.439969Z","iopub.execute_input":"2025-10-13T01:04:33.440184Z","iopub.status.idle":"2025-10-13T01:04:33.455931Z","shell.execute_reply.started":"2025-10-13T01:04:33.440161Z","shell.execute_reply":"2025-10-13T01:04:33.455395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pydicom\n\ndef quick_sort_dicom_files(dicom_file_list):\n    print(\"quick sort dicom files\")\n    sorting_data = []\n    \n    for dicom_file in dicom_file_list:\n        try:\n            dicom_obj = pydicom.dcmread(dicom_file, stop_before_pixels=True, force=True)\n            inst_num = getattr(dicom_obj, 'InstanceNumber', None)\n            img_position = getattr(dicom_obj, 'ImagePositionPatient', [None, None, None])\n            z_coord = img_position[2] if img_position and len(img_position) == 3 else None\n\n            if inst_num is not None:\n                sorting_data.append((int(inst_num), 0, dicom_file))\n            elif z_coord is not None:\n                sorting_data.append((float('inf'), float(z_coord), dicom_file))\n            else:\n                sorting_data.append((float('inf'), float('inf'), dicom_file))\n        except Exception:\n            sorting_data.append((float('inf'), float('inf'), dicom_file))\n    \n    sorting_data.sort() \n    return [item[2] for item in sorting_data]\n\ndef arrange_series(series_data):\n    print(\"arrange_series\")\n    series_identifier, dicom_files = series_data\n    ordered_files = quick_sort_dicom_files(dicom_files)\n    return series_identifier, ordered_files\n\n\ndef apply_windowing_transform(image_array, center_val, width_val):\n    print(\"apply_windowing_transform\")\n    \"\"\"\n    Perform DICOM windowing transformation on the input image array.\n    Windowing enhances the visualization of particular anatomical regions \n    by adjusting the pixel intensity display range.\n\n    Parameters\n    ----------\n    image_array : np.ndarray\n        Input DICOM image (2D grayscale array).\n    center_val : float\n        The midpoint of the intensity range to display.\n    width_val : float\n        The range width of the intensity values to display.\n\n    Returns\n    -------\n    np.ndarray\n        The windowed image array scaled to 8-bit range (0–255).\n    \"\"\"\n    min_val = center_val - width_val // 2\n    max_val = center_val + width_val // 2\n\n    image_array = np.clip(image_array, min_val, max_val)\n    image_array = (image_array - min_val) / (max_val - min_val + 1e-7)\n    \n    return (image_array * 255).astype(np.uint8)\n\n\ndef fetch_windowing_values(modality_type):\n    print(\"fetch_windowing_values\")\n    \"\"\"\n    Retrieve default window center and width based on the imaging modality.\n\n    Parameters\n    ----------\n    modality_type : str\n        The type of DICOM imaging modality (e.g., 'CT', 'MRI', 'MRA', etc.)\n\n    Returns\n    -------\n    Tuple[float, float]\n        (window_center, window_width)\n    \"\"\"\n    modality_windows = {\n        'CT': (40, 80),\n        'CTA': (50, 350),\n        'MRA': (600, 1200),\n        'MRI': (40, 80),\n        'MRI T2': (40, 80),\n        'MRI T1post': (40, 80),\n    }\n    return modality_windows.get(modality_type, (40, 80))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:33.456639Z","iopub.execute_input":"2025-10-13T01:04:33.456813Z","iopub.status.idle":"2025-10-13T01:04:33.474511Z","shell.execute_reply.started":"2025-10-13T01:04:33.4568Z","shell.execute_reply":"2025-10-13T01:04:33.474Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Showing How the Sorting is performed**","metadata":{}},{"cell_type":"code","source":"d1 = pydicom.dcmread(\"/kaggle/input/rsna-intracranial-aneurysm-detection/series/1.2.826.0.1.3680043.8.498.10004044428023505108375152878107656647/1.2.826.0.1.3680043.8.498.56949904638593632206697234148881166799.dcm\")\nd2 = pydicom.dcmread(\"/kaggle/input/rsna-intracranial-aneurysm-detection/series/1.2.826.0.1.3680043.8.498.10004044428023505108375152878107656647/1.2.826.0.1.3680043.8.498.12396711188070994245238798082430967707.dcm\", stop_before_pixels=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:33.475317Z","iopub.execute_input":"2025-10-13T01:04:33.475559Z","iopub.status.idle":"2025-10-13T01:04:33.498211Z","shell.execute_reply.started":"2025-10-13T01:04:33.475538Z","shell.execute_reply":"2025-10-13T01:04:33.497724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"d1.PhotometricInterpretation","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:33.500775Z","iopub.execute_input":"2025-10-13T01:04:33.500961Z","iopub.status.idle":"2025-10-13T01:04:33.505564Z","shell.execute_reply.started":"2025-10-13T01:04:33.500947Z","shell.execute_reply":"2025-10-13T01:04:33.50503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"d1[\"InstanceNumber\"], d1[\"ImagePositionPatient\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:33.506172Z","iopub.execute_input":"2025-10-13T01:04:33.506366Z","iopub.status.idle":"2025-10-13T01:04:33.520798Z","shell.execute_reply.started":"2025-10-13T01:04:33.506352Z","shell.execute_reply":"2025-10-13T01:04:33.52008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"d2[\"InstanceNumber\"], d2[\"ImagePositionPatient\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:33.521546Z","iopub.execute_input":"2025-10-13T01:04:33.522056Z","iopub.status.idle":"2025-10-13T01:04:33.534877Z","shell.execute_reply.started":"2025-10-13T01:04:33.522031Z","shell.execute_reply":"2025-10-13T01:04:33.534108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# === Load train.csv ===\ndf_train = pd.read_csv('/kaggle/input/rsna-intracranial-aneurysm-detection/train.csv')\nseries_uids_full = df_train['SeriesInstanceUID'].unique()\n\nif DEBUG:\n    print(\"Using only first 10 series_uids\")\n    series_uids = series_uids_full[:10]\nelse:\n    print(\"Using all series_uids\")\n    series_uids = series_uids_full\n\n# === Build map: SeriesInstanceUID → list of DICOM paths ===\n\nseries_dicom_map = {\n    suid: glob.glob(os.path.join(DATA_ROOT, 'series', suid, '*.dcm'))\n    for suid in series_uids\n}\n\n\nwith Pool(cpu_count()) as pool:\n    sorted_results = list(pool.imap(arrange_series, series_dicom_map.items()))\n\nrows = []\nfor series_uid, sorted_paths in sorted_results:\n    modality = df_train[df_train['SeriesInstanceUID'] == series_uid]['Modality'].iloc[0]\n    for idx, path in enumerate(sorted_paths):\n        sop_uid = os.path.splitext(os.path.basename(path))[0]\n        rows.append({\n            'SeriesInstanceUID': series_uid,\n            'SOPInstanceUID': sop_uid,\n            'dicom_filename': path,\n            'relative_index': idx,\n            'Modality': modality\n        })\n\n\n# === Save as DataFrame ===\ndf_series_index_mapping = pd.DataFrame(rows)\ndf_series_index_mapping = df_series_index_mapping.sort_values(\n    by=['SeriesInstanceUID', 'relative_index']\n)\ndf_series_index_mapping.to_csv('series_index_mapping.csv', index=False)\nprint(\"Saved series_index_mapping.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:33.535652Z","iopub.execute_input":"2025-10-13T01:04:33.535857Z","iopub.status.idle":"2025-10-13T01:04:35.813018Z","shell.execute_reply.started":"2025-10-13T01:04:33.535842Z","shell.execute_reply":"2025-10-13T01:04:35.812138Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Cache files sorting in each series for fast retrieval ","metadata":{}},{"cell_type":"markdown","source":"# Explore csv","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(f'{DATA_ROOT}/train.csv')\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:35.814143Z","iopub.execute_input":"2025-10-13T01:04:35.814447Z","iopub.status.idle":"2025-10-13T01:04:35.841202Z","shell.execute_reply.started":"2025-10-13T01:04:35.814412Z","shell.execute_reply":"2025-10-13T01:04:35.840568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:35.841869Z","iopub.execute_input":"2025-10-13T01:04:35.842063Z","iopub.status.idle":"2025-10-13T01:04:35.847056Z","shell.execute_reply.started":"2025-10-13T01:04:35.842046Z","shell.execute_reply":"2025-10-13T01:04:35.846316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['SeriesInstanceUID'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:35.847847Z","iopub.execute_input":"2025-10-13T01:04:35.848103Z","iopub.status.idle":"2025-10-13T01:04:35.866205Z","shell.execute_reply.started":"2025-10-13T01:04:35.848081Z","shell.execute_reply":"2025-10-13T01:04:35.865399Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"dicom.PhotometricInterpretation shows 'MONOCHROME1'\n\n'MONOCHROME2'\n\n'RGB'\n\n'YBR_FULL'","metadata":{}},{"cell_type":"markdown","source":"# Export png from dcm","metadata":{}},{"cell_type":"code","source":"def dicom_to_png(src_path, dst_path, width=224, to_rgb=False, apply_windowing=False, modality='CT'):\n    print(\"dicom_to_png\")\n    try:\n        dicom = pydicom.dcmread(src_path)\n        tsuid = dicom.file_meta.TransferSyntaxUID  \n        img = dicom.pixel_array\n        interp = dicom.PhotometricInterpretation\n\n        if interp == \"YBR_FULL\":\n            img = convert_color_space(img, 'YBR_FULL', 'RGB')\n\n        if img.ndim == 3:\n            if interp in [\"RGB\", \"YBR_FULL\"]:\n                img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n            elif img.shape[2] == 1:\n                img = img[:, :, 0]\n\n        if apply_windowing:\n            window_center, window_width = fetch_windowing_values(modality)\n            img = apply_windowing_transform(img, window_center, window_width)\n\n        img = img.astype(np.float32)\n        img_min, img_max = img.min(), img.max()\n        if img_max > img_min:\n            img = (img - img_min) / (img_max - img_min)\n        else:\n            img[:] = 0 \n\n        if interp == \"MONOCHROME1\":\n            img = 1 - img\n\n        img = (img * 255).astype(np.uint8)\n\n        if img is None or img.size == 0:\n            print(f\"Invalid image data in: {src_path}\")\n            return\n\n        try:\n            img = cv2.resize(img, (width, width))\n        except Exception as e:\n            print(f\"[ERROR] Resize failed: {src_path} | {e}\")\n            return\n\n        if to_rgb:\n            img = cv2.cvtColor(img, cv2.COLOR_GRAY2RGB)\n\n        os.makedirs(os.path.dirname(dst_path), exist_ok=True)\n\n        cv2.imwrite(dst_path, img)\n        # print(f\"[OK] Saved: {dst_path}\")\n\n    except Exception as e:\n        print(f\"[ERROR] Failed to process {src_path}: {e}\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:35.867125Z","iopub.execute_input":"2025-10-13T01:04:35.867387Z","iopub.status.idle":"2025-10-13T01:04:35.88402Z","shell.execute_reply.started":"2025-10-13T01:04:35.867368Z","shell.execute_reply":"2025-10-13T01:04:35.883378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nexclude_cols = ['SeriesInstanceUID', 'PatientAge', 'PatientSex', 'Modality', 'Aneurysm Present']\n\n# Get list of columns excluding the specified ones\nlocation = [col for col in df.columns if col not in exclude_cols]\n\nprint(location)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:35.884826Z","iopub.execute_input":"2025-10-13T01:04:35.88503Z","iopub.status.idle":"2025-10-13T01:04:35.902488Z","shell.execute_reply.started":"2025-10-13T01:04:35.885014Z","shell.execute_reply":"2025-10-13T01:04:35.901922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"seriesInst_ids = df['SeriesInstanceUID'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:35.903122Z","iopub.execute_input":"2025-10-13T01:04:35.903326Z","iopub.status.idle":"2025-10-13T01:04:35.916836Z","shell.execute_reply.started":"2025-10-13T01:04:35.903302Z","shell.execute_reply":"2025-10-13T01:04:35.916243Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pdf = df[df['SeriesInstanceUID'] == \"1.2.826.0.1.3680043.8.498.10188636688783982623025997809119805350\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:35.917539Z","iopub.execute_input":"2025-10-13T01:04:35.918258Z","iopub.status.idle":"2025-10-13T01:04:35.929975Z","shell.execute_reply.started":"2025-10-13T01:04:35.918237Z","shell.execute_reply":"2025-10-13T01:04:35.929318Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Check the maximum number of input DICOM files per series directory to decide whether {j:03d}.png is sufficient or if {j:04d}.png is needed.","metadata":{}},{"cell_type":"code","source":"for _, row in pdf.iterrows():\n    print(row)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:35.930696Z","iopub.execute_input":"2025-10-13T01:04:35.9309Z","iopub.status.idle":"2025-10-13T01:04:35.946778Z","shell.execute_reply.started":"2025-10-13T01:04:35.930877Z","shell.execute_reply":"2025-10-13T01:04:35.946144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_mapping = pd.read_csv('series_index_mapping.csv')\n\noutputList = []\n\nif DEBUG:\n    series_subset = seriesInst_ids[:5]\n    print(\"Processing first 5 SeriesInstanceUIDs\")\nelse:\n    series_subset = seriesInst_ids\n    print(\"Processing all SeriesInstanceUIDs\")\n\nfor idx, si in enumerate(series_subset):\n    print(f\"Processing series {idx + 1}/{len(series_subset)}: {si}\") \n    \n    pdf = df[df['SeriesInstanceUID'] == si]\n    \n    for _, row in pdf.iterrows():\n        location_exist = [col for col in location if row[col] == 1]\n        \n        if not location_exist:\n            continue \n\n        for loca in location_exist:\n            new_location = loca.replace('/', '_')  \n            \n            df_series = df_mapping[df_mapping['SeriesInstanceUID'] == si]\n            df_series = df_series.sort_values(by='relative_index')\n\n            if df_series.empty:\n                continue\n\n            # Create output directory\n            out_dir = f'dcm_png/{new_location}/{si}'\n            os.makedirs(out_dir, exist_ok=True)\n\n            for j, row_mapping in enumerate(df_series.itertuples(index=False)):\n                impath = row_mapping.dicom_filename\n                dst = f'{out_dir}/{j:04d}.png'\n                outputList.append({\n                    'impath': impath,\n                    'dst': dst,\n                    'modality': row_mapping.Modality\n                })\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:35.947399Z","iopub.execute_input":"2025-10-13T01:04:35.947669Z","iopub.status.idle":"2025-10-13T01:04:35.986092Z","shell.execute_reply.started":"2025-10-13T01:04:35.947648Z","shell.execute_reply":"2025-10-13T01:04:35.985099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"location_exist","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:35.986985Z","iopub.execute_input":"2025-10-13T01:04:35.987267Z","iopub.status.idle":"2025-10-13T01:04:35.991638Z","shell.execute_reply.started":"2025-10-13T01:04:35.987244Z","shell.execute_reply":"2025-10-13T01:04:35.990958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print (\"Length of outputList : %d\" % len (outputList))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:35.992446Z","iopub.execute_input":"2025-10-13T01:04:35.992701Z","iopub.status.idle":"2025-10-13T01:04:36.006106Z","shell.execute_reply.started":"2025-10-13T01:04:35.992681Z","shell.execute_reply":"2025-10-13T01:04:36.005552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"outputList[:2]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:36.006968Z","iopub.execute_input":"2025-10-13T01:04:36.007214Z","iopub.status.idle":"2025-10-13T01:04:36.02224Z","shell.execute_reply.started":"2025-10-13T01:04:36.007192Z","shell.execute_reply":"2025-10-13T01:04:36.021531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for img in outputList[:3]:\n    print(img)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:36.022918Z","iopub.execute_input":"2025-10-13T01:04:36.023196Z","iopub.status.idle":"2025-10-13T01:04:36.035617Z","shell.execute_reply.started":"2025-10-13T01:04:36.02318Z","shell.execute_reply":"2025-10-13T01:04:36.034892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mapping_dict = {\n    (row['SeriesInstanceUID'], row['SOPInstanceUID']): (row['relative_index'], row['dicom_filename'])\n    for _, row in df_mapping.iterrows()\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:36.036372Z","iopub.execute_input":"2025-10-13T01:04:36.036634Z","iopub.status.idle":"2025-10-13T01:04:36.164731Z","shell.execute_reply.started":"2025-10-13T01:04:36.036612Z","shell.execute_reply":"2025-10-13T01:04:36.163905Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Create relative infos for the converted png","metadata":{}},{"cell_type":"code","source":"df_localizers = pd.read_csv(f'{DATA_ROOT}/train_localizers.csv')\ndf_localizers.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:36.168037Z","iopub.execute_input":"2025-10-13T01:04:36.168248Z","iopub.status.idle":"2025-10-13T01:04:36.18438Z","shell.execute_reply.started":"2025-10-13T01:04:36.168231Z","shell.execute_reply":"2025-10-13T01:04:36.183849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mapping_dict = {\n    (row['SeriesInstanceUID'], row['SOPInstanceUID']): (row['relative_index'], row['dicom_filename'])\n    for _, row in df_mapping.iterrows()\n}\n\nif DEBUG:\n    print(\"Using only first 5 rows of df_localizers\")\n    df_coords = df_localizers.iloc[:5].copy()\nelse:\n    print(\"Using full df_localizers\")\n    df_coords = df_localizers.copy()\n\n# Output columns\nrelative_indices = []\nrelative_xs = []\nrelative_ys = []\n\nfor idx, row in df_coords.iterrows():\n    key = (row['SeriesInstanceUID'], row['SOPInstanceUID'])\n\n    if key not in mapping_dict:\n        relative_indices.append(None)\n        relative_xs.append(None)\n        relative_ys.append(None)\n        continue\n\n    relative_index, dicom_path = mapping_dict[key]\n    relative_indices.append(relative_index)\n\n    try:\n        ds = pydicom.dcmread(dicom_path, stop_before_pixels=True)\n        h, w = int(ds.Rows), int(ds.Columns)\n\n        coords = row['coordinates']\n        if isinstance(coords, str):\n            coords = eval(coords)\n\n        x_rel = (coords['x'] / w) * GLOBAL_WIDTH\n        y_rel = (coords['y'] / h) * GLOBAL_WIDTH\n\n        relative_xs.append(x_rel)\n        relative_ys.append(y_rel)\n\n    except Exception:\n        relative_xs.append(None)\n        relative_ys.append(None)\n\n# Add columns back to DataFrame\ndf_coords['relative_index'] = relative_indices\ndf_coords['relative_x'] = relative_xs\ndf_coords['relative_y'] = relative_ys\n\noutput_csv_path = 'train_localizers_with_relative.csv'\ndf_coords.to_csv(output_csv_path, index=False)\nprint(f\"Saved updated DataFrame with relative info to: {output_csv_path}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:36.185036Z","iopub.execute_input":"2025-10-13T01:04:36.185218Z","iopub.status.idle":"2025-10-13T01:04:36.318546Z","shell.execute_reply.started":"2025-10-13T01:04:36.185195Z","shell.execute_reply":"2025-10-13T01:04:36.318006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"next(df_coords.iterrows())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:36.319211Z","iopub.execute_input":"2025-10-13T01:04:36.319389Z","iopub.status.idle":"2025-10-13T01:04:36.324837Z","shell.execute_reply.started":"2025-10-13T01:04:36.319374Z","shell.execute_reply":"2025-10-13T01:04:36.324151Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Save the PNG from DICOM","metadata":{}},{"cell_type":"code","source":"from multiprocessing import cpu_count\nn_cores = cpu_count()\nprint(f'Number of Logical CPU cores: {n_cores}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:36.325548Z","iopub.execute_input":"2025-10-13T01:04:36.326169Z","iopub.status.idle":"2025-10-13T01:04:36.339243Z","shell.execute_reply.started":"2025-10-13T01:04:36.326151Z","shell.execute_reply":"2025-10-13T01:04:36.338554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# start_time = time.time()\n\n# Parallel(n_jobs=n_cores)(\n#     delayed(dicom_to_png)(img['impath'], img['dst'], GLOBAL_WIDTH, apply_windowing=True, modality=img['modality'])\n#     for img in outputList\n# )\n\n# elapsed = time.time() - start_time\n\n# # Format time nicely\n# hours, rem = divmod(elapsed, 3600)\n# minutes, seconds = divmod(rem, 60)\n# print(f\"Total running time: {int(hours)}h {int(minutes)}m {seconds:.2f}s\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:36.340097Z","iopub.execute_input":"2025-10-13T01:04:36.340362Z","iopub.status.idle":"2025-10-13T01:04:36.35393Z","shell.execute_reply.started":"2025-10-13T01:04:36.340338Z","shell.execute_reply":"2025-10-13T01:04:36.353259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def convert_all_dicoms(output_list, num_cores):\n    print(f\"Starting DICOM to PNG conversion using {num_cores} cores...\")\n    start = time.time()\n\n    # Parallel conversion\n    Parallel(n_jobs=num_cores)(\n        delayed(dicom_to_png)(\n            img['impath'], \n            img['dst'], \n            GLOBAL_WIDTH, \n            apply_windowing=True, \n            modality=img['modality']\n        )\n        for img in output_list\n    )\n\n    # Calculate elapsed time\n    total_time = time.time() - start\n    hrs, rem = divmod(total_time, 3600)\n    mins, secs = divmod(rem, 60)\n\n    print(f\"✅ Conversion complete in {int(hrs)}h {int(mins)}m {secs:.2f}s\")\n\n# Usage\nconvert_all_dicoms(outputList, n_cores)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:36.354777Z","iopub.execute_input":"2025-10-13T01:04:36.354995Z","iopub.status.idle":"2025-10-13T01:04:39.705221Z","shell.execute_reply.started":"2025-10-13T01:04:36.35498Z","shell.execute_reply":"2025-10-13T01:04:39.704618Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Visualization of the Dicom and converted PNG with useful information","metadata":{}},{"cell_type":"code","source":"\n# def show_dicom_and_png_with_info(row, df_meta, rd, box_size_resized=10):\n#     print(\"show_dicom_and_png_with_info\")\n\n#     # Load the series_index_mapping.csv once\n#     df_index_map = pd.read_csv(\"/kaggle/working/series_index_mapping.csv\")\n\n#     # Parse identifiers\n#     series_uid = row['SeriesInstanceUID']\n#     sop_uid = row['SOPInstanceUID']\n#     rel_index = row['relative_index']\n#     rel_x = row['relative_x']\n#     rel_y = row['relative_y']\n#     loca = row['location'].replace('/', '_')\n\n#     # Get metadata row\n#     df_row = df_meta[df_meta['SeriesInstanceUID'] == series_uid].iloc[0]\n#     patient_age = df_row['PatientAge']\n#     patient_sex = df_row['PatientSex']\n#     modality = df_row['Modality']\n\n#     # Aneurysm location columns\n#     aneurysm_location_cols = [\n#         \"Left Infraclinoid Internal Carotid Artery\", \"Right Infraclinoid Internal Carotid Artery\",\n#         \"Left Supraclinoid Internal Carotid Artery\", \"Right Supraclinoid Internal Carotid Artery\",\n#         \"Left Middle Cerebral Artery\", \"Right Middle Cerebral Artery\",\n#         \"Anterior Communicating Artery\", \"Left Anterior Cerebral Artery\",\n#         \"Right Anterior Cerebral Artery\", \"Left Posterior Communicating Artery\",\n#         \"Right Posterior Communicating Artery\", \"Basilar Tip\",\n#         \"Other Posterior Circulation\"\n#     ]\n#     aneurysm_locations = [col for col in aneurysm_location_cols if df_row.get(col, 0) == 1]\n\n#     # === Load PNG ===\n#     rel_index_int = int(rel_index)\n#     png_path = f'dcm_png/{loca}/{series_uid}/{rel_index_int:04d}.png'\n#     img_png = cv2.imread(png_path)\n#     if img_png is None:\n#         print(f\"[!] PNG not found: {png_path}\")\n#         return\n#     img_png = cv2.cvtColor(img_png, cv2.COLOR_BGR2RGB)\n\n#     # Draw box on PNG\n#     x1_png = int(rel_x - box_size_resized / 2)\n#     y1_png = int(rel_y - box_size_resized / 2)\n#     x2_png = int(rel_x + box_size_resized / 2)\n#     y2_png = int(rel_y + box_size_resized / 2)\n#     cv2.rectangle(img_png, (x1_png, y1_png), (x2_png, y2_png), (255, 0, 0), 2)\n\n#     # === Load DICOM using CSV ===\n#     df_series = df_index_map[df_index_map['SeriesInstanceUID'] == series_uid]\n#     dicom_path_row = df_series[df_series['SOPInstanceUID'] == sop_uid]\n#     if dicom_path_row.empty:\n#         print(f\"[!] DICOM path not found in CSV for SOPInstanceUID: {sop_uid}\")\n#         return\n\n#     dicom_path = dicom_path_row['dicom_filename'].values[0]\n\n#     try:\n#         ds = pydicom.dcmread(dicom_path)\n#         img_dcm = ds.pixel_array.astype(np.float32)\n#         img_dcm = (img_dcm - img_dcm.min()) / (img_dcm.max() - img_dcm.min() + 1e-6)\n#         img_dcm = (img_dcm * 255).astype(np.uint8)\n#         img_dcm_rgb = cv2.cvtColor(img_dcm, cv2.COLOR_GRAY2RGB)\n#     except Exception as e:\n#         print(f\"[!] Error reading DICOM: {dicom_path} | {e}\")\n#         return\n\n#     # Draw box on DICOM\n#     coords = row['coordinates']\n#     if isinstance(coords, str):\n#         coords = eval(coords)\n#     x_orig = int(coords['x'])\n#     y_orig = int(coords['y'])\n\n#     h_dcm, w_dcm = img_dcm.shape\n#     scale_x = w_dcm / GLOBAL_WIDTH\n#     scale_y = h_dcm / GLOBAL_WIDTH\n#     box_size_dcm_x = int(box_size_resized * scale_x)\n#     box_size_dcm_y = int(box_size_resized * scale_y)\n\n#     x1_dcm = max(0, int(x_orig - box_size_dcm_x / 2))\n#     y1_dcm = max(0, int(y_orig - box_size_dcm_y / 2))\n#     x2_dcm = min(w_dcm - 1, int(x_orig + box_size_dcm_x / 2))\n#     y2_dcm = min(h_dcm - 1, int(y_orig + box_size_dcm_y / 2))\n#     cv2.rectangle(img_dcm_rgb, (x1_dcm, y1_dcm), (x2_dcm, y2_dcm), (255, 0, 0), 2)\n\n#     # === Plot ===\n#     fig, axes = plt.subplots(1, 2, figsize=(12, 6))\n#     fig.subplots_adjust(top=0.75)\n\n#     info_line = f\"Modality: {modality} | Age: {patient_age} | Sex: {patient_sex}\"\n#     aneurysm_line = (\n#         \"Aneurysm Present in: \" + \", \".join(aneurysm_locations)\n#         if aneurysm_locations else \"No Aneurysm Present\"\n#     )\n#     tech_line = f\"Series UID: {series_uid}\\nLocation: {loca} | Frame Index: {rel_index}\"\n\n#     fig.suptitle(\n#         f\"{info_line}\\n{aneurysm_line}\\n{tech_line}\",\n#         fontsize=12, y=1.1, ha='center'\n#     )\n\n#     axes[0].imshow(img_dcm_rgb)\n#     axes[0].set_title(\"Original DICOM\")\n#     axes[0].set_xlabel(\"Pixels\")\n#     axes[0].set_ylabel(\"Pixels\")\n#     axes[0].grid(True, linestyle='--', linewidth=0.5)\n\n#     axes[1].imshow(img_png)\n#     axes[1].set_title(f\"Converted & Resized PNG ({GLOBAL_WIDTH}x{GLOBAL_WIDTH})\")\n#     axes[1].set_xlabel(\"Pixels\")\n#     axes[1].set_ylabel(\"Pixels\")\n#     axes[1].grid(True, linestyle='--', linewidth=0.5)\n\n#     plt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:39.70586Z","iopub.execute_input":"2025-10-13T01:04:39.706486Z","iopub.status.idle":"2025-10-13T01:04:39.71115Z","shell.execute_reply.started":"2025-10-13T01:04:39.706466Z","shell.execute_reply":"2025-10-13T01:04:39.710656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def converted_png(row):\n    \n    series_uid = row['SeriesInstanceUID']\n    rel_index = int(row['relative_index'])\n    rel_x, rel_y = row['relative_x'], row['relative_y']\n    location = row['location'].replace('/', '_')\n\n    # Define paths\n    png_dir = f'/kaggle/working/dcm_png/{location}/{series_uid}'\n    png_path = os.path.join(png_dir, f\"{rel_index:04d}.png\")\n\n    if not os.path.exists(png_path):\n        print(f\"PNG not found: {png_path}\")\n        return None\n\n    img = cv2.imread(png_path)\n    if img is None:\n        print(f\"Failed to read PNG: {png_path}\")\n        return None\n\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    shutil.make_archive(png_path, 'zip', png_dir)\n    print(f\"📦 Zipped all PNGs for {series_uid} → {zip_output_path}.zip\")\n\n    return boxed_path\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:39.711957Z","iopub.execute_input":"2025-10-13T01:04:39.712197Z","iopub.status.idle":"2025-10-13T01:04:39.731233Z","shell.execute_reply.started":"2025-10-13T01:04:39.712176Z","shell.execute_reply":"2025-10-13T01:04:39.730631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport zipfile\n\ndef converted_png(row):\n    series_uid = row['SeriesInstanceUID']\n    rel_index = int(row['relative_index'])\n    rel_x, rel_y = row['relative_x'], row['relative_y']\n    location = row['location'].replace('/', '_')\n\n    # Define paths\n    png_dir = f'/kaggle/working/dcm_png/{location}/{series_uid}'\n    png_path = os.path.join(png_dir, f\"{rel_index:04d}.png\")\n\n    # Check if PNG exists\n    if not os.path.exists(png_path):\n        print(f\"⚠️ PNG not found: {png_path}\")\n        return None\n\n    img = cv2.imread(png_path)\n    if img is None:\n        print(f\"⚠️ Failed to read PNG: {png_path}\")\n        return None\n\n    # Convert color (just reading for validation)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    # Define what to zip\n    root_dir_to_zip = '/kaggle/working/dcm_png'\n    zip_output_path = '/kaggle/working/dcm_png_zip_strong.zip'\n\n    # Strong compression\n    with zipfile.ZipFile(zip_output_path, 'w', compression=zipfile.ZIP_DEFLATED, compresslevel=9) as zipf:\n        for foldername, subfolders, filenames in os.walk(root_dir_to_zip):\n            for filename in filenames:\n                file_path = os.path.join(foldername, filename)\n                arcname = os.path.relpath(file_path, root_dir_to_zip)\n                zipf.write(file_path, arcname)\n\n    print(f\"📦 Entire directory compressed (strong) → {zip_output_path}\")\n    return zip_output_path\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:39.731957Z","iopub.execute_input":"2025-10-13T01:04:39.73213Z","iopub.status.idle":"2025-10-13T01:04:39.751589Z","shell.execute_reply.started":"2025-10-13T01:04:39.732116Z","shell.execute_reply":"2025-10-13T01:04:39.751041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfor i in range(3):\n    converted_png(df_coords.iloc[i])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:04:39.752299Z","iopub.execute_input":"2025-10-13T01:04:39.752505Z","iopub.status.idle":"2025-10-13T01:09:19.879774Z","shell.execute_reply.started":"2025-10-13T01:04:39.75249Z","shell.execute_reply":"2025-10-13T01:09:19.878794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q pydrive2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:18:06.112749Z","iopub.execute_input":"2025-10-13T01:18:06.113342Z","iopub.status.idle":"2025-10-13T01:18:10.72771Z","shell.execute_reply.started":"2025-10-13T01:18:06.113316Z","shell.execute_reply":"2025-10-13T01:18:10.726757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pydrive2.auth import GoogleAuth\nfrom pydrive2.drive import GoogleDrive\n\ngauth = GoogleAuth()\ngauth.LocalWebserverAuth()  # Opens a Google login page — sign in and allow access\ndrive = GoogleDrive(gauth)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:18:22.853347Z","iopub.execute_input":"2025-10-13T01:18:22.853671Z","iopub.status.idle":"2025-10-13T01:18:23.716848Z","shell.execute_reply.started":"2025-10-13T01:18:22.853644Z","shell.execute_reply":"2025-10-13T01:18:23.715982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from IPython.display import FileLink\n\n# # Display a clickable download link in the notebook\n# FileLink('/kaggle/working/dcm_png_zip.zip')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:09:19.880617Z","iopub.execute_input":"2025-10-13T01:09:19.880908Z","iopub.status.idle":"2025-10-13T01:09:19.884447Z","shell.execute_reply.started":"2025-10-13T01:09:19.880881Z","shell.execute_reply":"2025-10-13T01:09:19.883657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# old_zip = '/kaggle/working/dcm_png_zip.zip'\n# if os.path.exists(old_zip):\n#     os.remove(old_zip)\n#     print(\"🧹 Deleted previous zip file.\")\n# else:\n#     print(\"No old zip found.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:09:19.88552Z","iopub.execute_input":"2025-10-13T01:09:19.885819Z","iopub.status.idle":"2025-10-13T01:09:20.766242Z","shell.execute_reply.started":"2025-10-13T01:09:19.885795Z","shell.execute_reply":"2025-10-13T01:09:20.765395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n\n# files_to_delete = [\n#     '/kaggle/working/dcm_png_part_aa',\n#     '/kaggle/working/dcm_png_part_ab'\n# ]\n\n# for file_path in files_to_delete:\n#     if os.path.exists(file_path):\n#         os.remove(file_path)\n#         print(f\"🧹 Deleted: {file_path}\")\n#     else:\n#         print(f\"⚠️ File not found: {file_path}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:09:20.767116Z","iopub.execute_input":"2025-10-13T01:09:20.767376Z","iopub.status.idle":"2025-10-13T01:09:21.045747Z","shell.execute_reply.started":"2025-10-13T01:09:20.767356Z","shell.execute_reply":"2025-10-13T01:09:21.04494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import zipfile\n# import os\n\n# root_dir = '/kaggle/working/dcm_png'\n# zip_path = '/kaggle/working/dcm_png_zip_strong.zip'\n\n# with zipfile.ZipFile(zip_path, 'w', compression=zipfile.ZIP_DEFLATED, compresslevel=9) as zipf:\n#     for foldername, subfolders, filenames in os.walk(root_dir):\n#         for filename in filenames:\n#             file_path = os.path.join(foldername, filename)\n#             arcname = os.path.relpath(file_path, root_dir)\n#             zipf.write(file_path, arcname)\n\n# print(f\"✅ Finished compressing → {zip_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:09:21.046538Z","iopub.execute_input":"2025-10-13T01:09:21.046791Z","iopub.status.idle":"2025-10-13T01:13:55.02589Z","shell.execute_reply.started":"2025-10-13T01:09:21.046774Z","shell.execute_reply":"2025-10-13T01:13:55.025122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# def concat_dicom_series_to_image(series_uid, location=None, df_coords=None, rd='.', \n#                                 resize_shape=(GLOBAL_WIDTH, GLOBAL_WIDTH), box_size_resized=10,\n#                                 layout='vertical', show=True, save_path=None,\n#                                 use_png=False):\n#     print(\"concat_dicom_series_to_image\")\n#     \"\"\"\n#     Concatenate all images in a DICOM series (or converted PNGs) into a single visual strip or grid.\n\n#     Args:\n#         series_uid (str): SeriesInstanceUID to visualize.\n#         location (str): Needed only if use_png=True. Disease location (for PNG folder path).\n#         df_coords (pd.DataFrame or None): Coordinate info for box overlays.\n#         rd (str): Root directory containing 'series' folder.\n#         resize_shape (tuple): Target (width, height) for resizing each image.\n#         box_size_resized (int): Box size (in resized image space).\n#         layout (str): 'vertical', 'horizontal', or 'grid'.\n#         show (bool): Whether to plot using matplotlib.\n#         save_path (str or None): If set, saves the output image.\n#         use_png (bool): If True, loads PNGs instead of DICOMs.\n\n#     Returns:\n#         None\n#     \"\"\"\n\n#     # Load index mapping from CSV\n#     df_index = pd.read_csv('/kaggle/working/series_index_mapping.csv')\n#     df_series = df_index[df_index['SeriesInstanceUID'] == series_uid]\n#     dicom_paths = df_series.sort_values(by='relative_index')['dicom_filename'].tolist()\n\n#     # Build coordinate map from df_coords\n#     coord_map = {}\n#     if df_coords is not None:\n#         df_sub = df_coords[df_coords['SeriesInstanceUID'] == series_uid]\n#         for _, row in df_sub.iterrows():\n#             sop = row['SOPInstanceUID']\n#             coords = row['coordinates']\n#             if isinstance(coords, str):\n#                 coords = eval(coords)\n#             coord_map[sop] = coords\n\n#     images = []\n\n#     if use_png:\n#         if location is None:\n#             raise ValueError(\"`location` must be provided when use_png=True\")\n\n#         png_dir = os.path.join('/kaggle/working/dcm_png', location.replace('/', '_'), series_uid)\n#         print(f\"Loading PNGs from: {png_dir}\")\n#         png_paths = sorted(glob.glob(os.path.join(png_dir, '*.png')))\n        \n#         if df_coords is not None:\n#             df_sub = df_coords[df_coords['SeriesInstanceUID'] == series_uid]\n\n#         for i, path in enumerate(png_paths):\n#             img = cv2.imread(path)\n#             if img is None:\n#                 continue\n#             img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n#             img_resized = cv2.resize(img, resize_shape)\n\n#             if df_coords is not None:\n#                 match = df_sub[df_sub['relative_index'] == i]\n#                 if not match.empty:\n#                     x = match.iloc[0]['relative_x']\n#                     y = match.iloc[0]['relative_y']\n#                     if pd.notna(x) and pd.notna(y):\n#                         x1 = int(x - box_size_resized / 2)\n#                         y1 = int(y - box_size_resized / 2)\n#                         x2 = int(x + box_size_resized / 2)\n#                         y2 = int(y + box_size_resized / 2)\n#                         cv2.rectangle(img_resized, (x1, y1), (x2, y2), (255, 0, 0), 2)\n\n#             images.append(img_resized)\n\n#     else:\n#         for path in dicom_paths:\n#             try:\n#                 ds = pydicom.dcmread(path)\n#                 img = ds.pixel_array.astype(np.float32)\n#                 img = (img - img.min()) / (img.max() - img.min() + 1e-6)\n#                 img = (img * 255).astype(np.uint8)\n#                 img_rgb = cv2.cvtColor(img, cv2.COLOR_GRAY2RGB)\n#                 img_resized = cv2.resize(img_rgb, resize_shape)\n\n#                 sop = ds.SOPInstanceUID\n#                 if sop in coord_map:\n#                     orig_h, orig_w = img.shape\n#                     x_orig = coord_map[sop]['x']\n#                     y_orig = coord_map[sop]['y']\n\n#                     # Scale original coords to resized image\n#                     x = int((x_orig / orig_w) * resize_shape[0])\n#                     y = int((y_orig / orig_h) * resize_shape[1])\n\n#                     x1 = int(x - box_size_resized / 2)\n#                     y1 = int(y - box_size_resized / 2)\n#                     x2 = int(x + box_size_resized / 2)\n#                     y2 = int(y + box_size_resized / 2)\n#                     cv2.rectangle(img_resized, (x1, y1), (x2, y2), (255, 0, 0), 2)\n\n#                 images.append(img_resized)\n#             except Exception as e:\n#                 print(f\"Failed to load DICOM: {path}: {e}\")\n#                 continue\n\n#     if not images:\n#         print(f\"[!] No images found for series: {series_uid}\")\n#         return None\n\n#     # Layout image\n#     if layout == 'vertical':\n#         final_image = cv2.vconcat(images)\n#     elif layout == 'horizontal':\n#         final_image = cv2.hconcat(images)\n#     elif layout == 'grid':\n#         grid_size = int(np.ceil(np.sqrt(len(images))))\n#         while len(images) < grid_size ** 2:\n#             images.append(np.zeros_like(images[0]))\n#         rows = [\n#             cv2.hconcat(images[i*grid_size:(i+1)*grid_size])\n#             for i in range(grid_size)\n#         ]\n#         final_image = cv2.vconcat(rows)\n#     else:\n#         raise ValueError(\"layout must be 'vertical', 'horizontal', or 'grid'\")\n\n#     if show:\n#         plt.figure(figsize=(12, 12))\n#         plt.imshow(final_image)\n#         plt.title(f\"Series UID: {series_uid} (using {'PNG' if use_png else 'DICOM'})\")\n#         plt.axis('off')\n#         plt.show()\n\n#     if save_path:\n#         cv2.imwrite(save_path, cv2.cvtColor(final_image, cv2.COLOR_RGB2BGR))\n#         print(f\"Saved: {save_path}\")\n\n#         if use_png:\n#             png_root = '/kaggle/working/dcm_png'\n#             zip_output_path = '/kaggle/working/dicom_pngs'\n\n#             # Make ZIP file of the directory where PNGs were loaded from\n#             shutil.make_archive(zip_output_path, 'zip', png_root)\n#             print(f\"All PNGs zipped at: {zip_output_path}.zip\")\n\n#     return\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:13:55.026813Z","iopub.execute_input":"2025-10-13T01:13:55.027096Z","iopub.status.idle":"2025-10-13T01:13:55.033573Z","shell.execute_reply.started":"2025-10-13T01:13:55.02707Z","shell.execute_reply":"2025-10-13T01:13:55.032909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# View entire series with annotation if coordinates available\n\n# concat_dicom_series_to_image(\n#     series_uid='1.2.826.0.1.3680043.8.498.10022796280698534221758473208024838831',\n#     df_coords=df_coords,\n#     rd=DATA_ROOT,\n#     show=True,\n#     use_png=False,\n#     layout='grid'\n# )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:13:55.03426Z","iopub.execute_input":"2025-10-13T01:13:55.034529Z","iopub.status.idle":"2025-10-13T01:13:55.054239Z","shell.execute_reply.started":"2025-10-13T01:13:55.034507Z","shell.execute_reply":"2025-10-13T01:13:55.053586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# concat_dicom_series_to_image(\n#     series_uid='1.2.826.0.1.3680043.8.498.10022796280698534221758473208024838831',\n#     df_coords=df_coords,\n#     rd=DATA_ROOT,\n#     show=True,\n#     location='Right Middle Cerebral Artery',\n#     use_png=True,\n#     layout='grid'\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T01:13:55.054928Z","iopub.execute_input":"2025-10-13T01:13:55.055148Z","iopub.status.idle":"2025-10-13T01:13:55.070305Z","shell.execute_reply.started":"2025-10-13T01:13:55.055124Z","shell.execute_reply":"2025-10-13T01:13:55.069644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}