{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":99552,"databundleVersionId":13851420,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Data-Pre-Processing for Training Acceleration","metadata":{}},{"cell_type":"markdown","source":"This notebook splits the original dataset into 4 to process the original raw data into the working volumes and builds a zip file which will later be 1/4 of the preprocessed volumes dataset. This approach was suggested by TA's and discussed on Piazza.\n\nINPUT: Needs the whole dataset's series folder and the train.csv file. It uses the SIDs of the patients to preprocess the volume (depending on the modality) and save the volume with the same name as the original folder that contained the dicom files of said patient with the same sid number","metadata":{}},{"cell_type":"markdown","source":"# Initialization","metadata":{}},{"cell_type":"code","source":"#Libraries from Demo\nimport os\nimport shutil\nfrom collections import defaultdict\n\nimport pandas as pd, ast\nimport polars as pl\nimport pydicom as dicom\n\nimport kaggle_evaluation.rsna_inference_server\n\n#Libraries from attempt\nimport torch\nimport random\nfrom torch.utils.data import Dataset, DataLoader\n\n#For visalization\nimport matplotlib.pyplot as plt\n\n#For ROI Extraction:\nimport numpy as np\nfrom scipy.ndimage import zoom as ndi_zoom\n\n#For Running Pre-Processing\nimport os, numpy as np, pandas as pd, multiprocessing as mp\nfrom tqdm.auto import tqdm\n\nID_COL = 'SeriesInstanceUID'\n\nLABEL_COLS = [\n    'Left Infraclinoid Internal Carotid Artery',\n    'Right Infraclinoid Internal Carotid Artery',\n    'Left Supraclinoid Internal Carotid Artery',\n    'Right Supraclinoid Internal Carotid Artery',\n    'Left Middle Cerebral Artery',\n    'Right Middle Cerebral Artery',\n    'Anterior Communicating Artery',\n    'Left Anterior Cerebral Artery',\n    'Right Anterior Cerebral Artery',\n    'Left Posterior Communicating Artery',\n    'Right Posterior Communicating Artery',\n    'Basilar Tip',\n    'Other Posterior Circulation',\n    'Aneurysm Present',\n]\n\n# All tags (other than PixelData and SeriesInstanceUID) that may be in a test set dcm file\nDICOM_TAG_ALLOWLIST = [\n    'BitsAllocated',\n    'BitsStored',\n    'Columns',\n    'FrameOfReferenceUID',\n    'HighBit',\n    'ImageOrientationPatient',\n    'ImagePositionPatient',\n    'InstanceNumber',\n    'Modality',\n    'PatientID',\n    'PhotometricInterpretation',\n    'PixelRepresentation',\n    'PixelSpacing',\n    'PlanarConfiguration',\n    'RescaleIntercept',\n    'RescaleSlope',\n    'RescaleType',\n    'Rows',\n    'SOPClassUID',\n    'SOPInstanceUID',\n    'SamplesPerPixel',\n    'SliceThickness',\n    'SpacingBetweenSlices',\n    'StudyInstanceUID',\n    'TransferSyntaxUID',\n]","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:05.042617Z","iopub.execute_input":"2025-12-06T22:48:05.042879Z","iopub.status.idle":"2025-12-06T22:48:14.604686Z","shell.execute_reply.started":"2025-12-06T22:48:05.042856Z","shell.execute_reply":"2025-12-06T22:48:14.603616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def seed_everything(seed=42):\n    \"\"\"\n    Set random seeds for reproducibility in deep learning projects.\n    \n    Args:\n        seed (int): Random seed value (default: 42)\n    \"\"\"\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)  # if using multi-GPU\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    os.environ['PYTHONHASHSEED'] = str(seed)\n\nseed_everything()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.605613Z","iopub.execute_input":"2025-12-06T22:48:14.606072Z","iopub.status.idle":"2025-12-06T22:48:14.619177Z","shell.execute_reply.started":"2025-12-06T22:48:14.606035Z","shell.execute_reply":"2025-12-06T22:48:14.617902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# training data's labels\ntrain_csv = \"/kaggle/input/rsna-intracranial-aneurysm-detection/train.csv\"\n#training data's localizable labels\nlocalizer_csv = \"/kaggle/input/rsna-intracranial-aneurysm-detection/train_localizers.csv\"\n#Path to the folder \"series, contains all the patients\"\nDICOM_PATH = \"/kaggle/input/rsna-intracranial-aneurysm-detection/series\"\n#Path to a patient's DICOM folder\ndicom_dir = \"/kaggle/input/rsna-intracranial-aneurysm-detection/series/1.2.826.0.1.3680043.8.498.10004044428023505108375152878107656647\"\n#Path to a patient's NII file (contains all slices already)\nnift_path = \"/kaggle/input/rsna-intracranial-aneurysm-detection/segmentations\"\n#Path to one slice DICOM file from one patient\ndicom_sample = \"/kaggle/input/rsna-intracranial-aneurysm-detection/series/1.2.826.0.1.3680043.8.498.10004044428023505108375152878107656647\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.623198Z","iopub.execute_input":"2025-12-06T22:48:14.623587Z","iopub.status.idle":"2025-12-06T22:48:14.646996Z","shell.execute_reply.started":"2025-12-06T22:48:14.623556Z","shell.execute_reply":"2025-12-06T22:48:14.645776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model configuration\nTARGET_SIZE = (160, 160, 160)  # Reduced size for memory efficiency\nTARGET_ISO_BRAIN = 1.25\nDEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\ntrain_df = pd.read_csv(train_csv)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.648087Z","iopub.execute_input":"2025-12-06T22:48:14.64842Z","iopub.status.idle":"2025-12-06T22:48:14.706397Z","shell.execute_reply.started":"2025-12-06T22:48:14.64839Z","shell.execute_reply":"2025-12-06T22:48:14.705355Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Supporting Functions\n","metadata":{}},{"cell_type":"markdown","source":"## Load the Localizers table with better preview (e.g., split coordinates into columns)","metadata":{}},{"cell_type":"code","source":"def _load_localizers_csv(path):\n    \"\"\"\n    INPUT: \n        path: path to the localizers csv file.\n    OUPTUT:\n        df: localizer table with updated columns, the lession localization was splitted into 2 columns.\n    \"\"\"\n    df = pd.read_csv(path)\n    # columns: SeriesInstanceUID, SOPInstanceUID, coordinates, location\n    df[\"coords\"] = df[\"coordinates\"].apply(lambda s: ast.literal_eval(s) if isinstance(s,str) else s)\n    df[\"x\"] = df[\"coords\"].apply(lambda d: float(d[\"x\"]))\n    df[\"y\"] = df[\"coords\"].apply(lambda d: float(d[\"y\"]))\n    return df[[\"SeriesInstanceUID\",\"SOPInstanceUID\",\"x\",\"y\",\"location\"]]\n\n\nloc_df = _load_localizers_csv(localizer_csv)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.707377Z","iopub.execute_input":"2025-12-06T22:48:14.70771Z","iopub.status.idle":"2025-12-06T22:48:14.805063Z","shell.execute_reply.started":"2025-12-06T22:48:14.70768Z","shell.execute_reply":"2025-12-06T22:48:14.804041Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Full Volume Pre-Processing","metadata":{}},{"cell_type":"code","source":"def _get_slice_order(slice_sample):\n    \"\"\"\n    INPUT: \n    slice_sample: dicom read single slice.\n    OUPTUT:\n    z_cord: slice's Z axis index for ordering.\n    \"\"\"\n    #check if the slice/dicom_file has ImageOrientationPatient\n    iop = getattr(slice_sample, \"ImageOrientationPatient\", None)\n    ipp = getattr(slice_sample, \"ImagePositionPatient\", None)\n\n    if iop is not None and ipp is not None and len(iop)>=6 and len(ipp)>=3:\n        x_cord = np.array(iop[:3], float)      # rows dir (x)\n        y_cord = np.array(iop[3:6], float)     # cols dir (y)\n        z_cord = np.cross(x_cord, y_cord)      # slice normal (z)\n        ipp = np.array(ipp[:3], float)\n        return float(np.dot(ipp, z_cord))\n\n    ins_num = getattr(slice_sample, \"InstanceNumber\", None)\n    if ins_num is not None:\n        z_cord = int(ins_num)\n        return z_cord \n\ndef _slice_by_slice_correct_rescale(list_of_slices):\n    \"\"\"\n    INPUT: \n    list_of_slices: List of dicom read slices (order does not matter).\n    OUPTUT:\n    corrected_slices: 1 channel, propper gray scale and corrected list of dicom slices in Array form.\n    \"\"\"\n    corrected_slices = []\n    for slce in list_of_slices:\n        slce_array = slce.pixel_array.astype(np.float32)\n        \n        #Fix the contrast of some images in case the smaller values are white instead of black for consistency\n        # https://dicom.innolitics.com/ciods/rt-dose/image-pixel/00280004\n        # https://dgobbi.github.io/vtk-dicom/doc/api/imageDisplay.html?utm_source=chatgpt.com\n        phot = str(getattr(slce, \"PhotometricInterpretation\", \"MONOCHROME2\")).upper()\n        if phot == \"MONOCHROME1\":\n            maxv = (2**int(getattr(slce, \"BitsStored\", arr.dtype.itemsize*8))) - 1\n            slce_array = maxv - slce_array\n\n        # https://collectiveminds.health/articles/the-ultimate-guide-to-preprocessing-medical-images-techniques-tools-and-best-practices-for-enhanced-diagnosis\n        #Collapse channels if there are more than 1, maybe just CT, but lets be sure.\n        spp = int(getattr(slce, \"SamplesPerPixel\", 1))\n        if spp == 3 and slce_array.ndim == 3:\n            slce_array = slce_array.mean(axis=-1)\n        \n        # Apply DICOM linear HU rescale because we are using different modalities: https://www.mathworks.com/help/medical-imaging/ug/overview-medical-image-preprocessing.html \n        slope = float(getattr(slce, \"RescaleSlope\", 1.0))\n        inter = float(getattr(slce, \"RescaleIntercept\", 0.0))\n        corrected_slices.append(slce_array * slope + inter)\n    return corrected_slices\n\ndef _get_slices_spacing(list_of_slices):\n    \"\"\"\n    INPUT: \n    list_of_slices: List of dicom read slices (order does not matter).\n    OUPTUT:\n    spacing_zyx: the pixel spacing from the dicom files in Array form\n    \"\"\"\n    sy, sx = map(float, getattr(list_of_slices[0], \"PixelSpacing\", [1.0,1.0]))\n    sz = None\n    \n    spacing = (sz, sy, sx)\n    st = getattr(list_of_slices[0], \"SliceThickness\", None)\n    sbs = getattr(list_of_slices[0], \"SpacingBetweenSlices\", None)\n    if st is not None:\n        sz = float(st)\n    if sbs is not None:\n        sz = float(sbs)\n\n    if sz is None:\n        iop = getattr(dicom_slices[0], \"ImageOrientationPatient\", None)\n        if iop is not None and len(iop) >= 6:\n            row = np.array(iop[:3], dtype=float)\n            col = np.array(iop[3:6], dtype=float)\n            norm   = np.cross(row, col)\n            scalars = []\n            for slc in list_of_slices:\n                ipp = getattr(slc, \"ImagePositionPatient\", None)\n                if ipp is not None and len(ipp)>=3:\n                    scalars.append(np.dot(np.array(ipp[:3], float), norm))\n            if len(scaalrs)>=2:\n                diffs = np.diff(np.sort(scalars))\n                sz = float(np.median(np.abs(diffs)))\n        if sz is None:\n            sz = 1.0\n    spacing_zyx = (sz, sy, sx)\n    return spacing_zyx\n\ndef load_dicom_series(dicom_patient_dir):\n    \"\"\"\n    INPUT: \n    dicom_patient_dir: Patient folder directory with dicom files per slice.\n    OUPTUT:\n    dicom_vol: Ordered and rescaled read dicom slices stacked to make a volume\n    modality: MRI, CTA, etc....\n\n    Variables Cheat Sheet\n    list_slices_files: The list of PATHS for each slice in the patient folder.\n    dicom_slices: dicom read files with data.\n    \"\"\"\n    list_slices_files = [\n        os.path.join(dicom_patient_dir, f) \n        for f in os.listdir(dicom_patient_dir) \n        if f.lower().endswith(\".dcm\")]\n    \n    if not list_slices_files: \n        raise ValueError(f\"No DICOM slice files found in patient folder: {dicom_patient_dir}.\")\n    \n    dicom_slices = []\n    for slice_file in list_slices_files:\n        dicom_slice = dicom.dcmread(\n            slice_file, stop_before_pixels=False, force=True)\n        dicom_slices.append(dicom_slice)\n\n    dicom_slices.sort(key=_get_slice_order)\n    modality = getattr(dicom_slices[0], 'Modality')\n    the_series_uid = getattr(dicom_slices[0], \"SeriesInstanceUID\")\n\n    # Calculate the Spacing using the dicom metadata of one slice.\n    spacing_zyx = _get_slices_spacing(dicom_slices)\n\n    \n    slices = _slice_by_slice_correct_rescale(dicom_slices)\n    dicom_vol = np.stack(slices, axis=0)\n\n    return modality, spacing_zyx, dicom_vol, the_series_uid, dicom_slices\n\n\n\ndef resample_isotropic(vol_zyx, spacing_zyx, target_iso=1, order=1):\n    \"\"\"\n    Resample a [Z,H,W] volume to isotropic spacing (target_iso in mm).\n    INPUT:\n    vol_zyx: np.ndarray [Z,H,W]\n    spacing_zyx: tuple (sz, sy, sx) in mm\n    target_iso: desired isotropic spacing in mm \n    order: interpolation order (1 = linear)\n    OUPTUT:\n    vol_iso: np.ndarray resampled volume\n    new_spacing_zyx: (target_iso, target_iso, target_iso)\n    \"\"\"\n    sz, sy, sx = map(float, spacing_zyx)\n    tz = ty = tx = float(target_iso)\n\n    # zoom factors are (old_spacing / new_spacing)\n    zoom_factors = (sz / tz, sy / ty, sx / tx)\n    vol_iso = ndi_zoom(vol_zyx, zoom=zoom_factors, order=order)\n    spacing_iso_zyx = (tz, ty, tx)\n    return spacing_iso_zyx, vol_iso.astype(np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.806237Z","iopub.execute_input":"2025-12-06T22:48:14.80653Z","iopub.status.idle":"2025-12-06T22:48:14.825483Z","shell.execute_reply.started":"2025-12-06T22:48:14.806509Z","shell.execute_reply":"2025-12-06T22:48:14.824331Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Modality Unification Pre-Processing","metadata":{}},{"cell_type":"code","source":"def modality_specific_preprocessing(vol, modality, cta_startegy=\"clip\"):\n    \"\"\"\n    Clip the extremes values of a volume (HU scaled)\n    INPUT:\n    volume: full sampled volume\n    OUPTUT: \n    volume: extremes clipped volume\n    Variables Cheat Sheet:\n    cta_startegy: clip or hu_window\n    \"\"\"\n    if modality == 'CTA' or modality == 'CT':\n        if cta_startegy == 'clip':\n            vol = _clip_by_percentile_per_volume(vol)\n        elif cta_startegy == 'hu_window':\n            vol = _apply_hu_window(vol)\n        elif cta_startegy == 'hu_range':\n            vol = _apply_hu_range(vol)\n        else:\n            vol = _clip_by_percentile_per_volume(vol)\n        \n        vol = _apply_normalization(vol, norm='min-max')\n        return vol\n    else:\n        vol = _apply_normalization(vol, norm='z-score')\n        return vol\n\ndef _clip_by_percentile_per_volume(vol, low_p=0.5, high_p=99.5):\n    \"\"\"\n    Clip the HU value of the CT volume\n    INPUT: \n    vol: volume tensor\n    low: lower percentile value to clip\n    high: higher percentile value to clip\n    \n    OUPTUT:\n        vol: clipped vol\n    \"\"\"\n    lo = np.percentile(vol, low_p)\n    hi = np.percentile(vol, high_p)\n    vol = np.clip(vol, lo, hi)\n    return ((vol - lo) / (hi - lo + 1e-6)).astype(np.float32)\n\ndef _apply_hu_window(vol, center=125, width=550):\n    \"\"\"\n    INPUT: \n    vol: volume tensor\n    center: center of the HU window\n    width: width of the HU unit window [center-width, center+width]\n    \n    OUPTUT:\n    vol: volume scaled to the defined window.\n    \"\"\"\n    lo = center - width/2.0\n    hi = center + width/2.0\n    vol = np.clip(vol, lo, hi)\n    return ((vol - lo) / (hi - lo + 1e-6)).astype(np.float32)\n\ndef _apply_hu_range(vol, hounsfield_range=(100, 600)):\n    \"\"\"\n    INPUT: \n    vol: volume tensor\n    center: center of the HU window\n    width: width of the HU unit window [center-width, center+width]\n    \n    OUPTUT:\n    range_vol: a mask of only the boders\n    \"\"\"\n    mask = (vol >= hounsfield_range[0]) & (vol <= hounsfield_range[1])\n    range_vol = np.where(mask, vol, 0)\n    return range_vol\n\ndef _apply_normalization(vol, norm='z-score'):\n    \"\"\"\n    INPUT: \n    vol: volume tenso\n    norm: normalization strategy (max-min for CTA, z-score for MRI)\n    \n    OUPTUT:\n    vol: Normalized volume tensor.\n    \"\"\"\n    if norm == 'min-max':\n        return (vol - vol.min()) / (vol.max() - vol.min() + 1e-8)\n    elif norm == 'z-score':\n        return (vol - vol.mean()) / vol.std()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.826621Z","iopub.execute_input":"2025-12-06T22:48:14.82698Z","iopub.status.idle":"2025-12-06T22:48:14.849889Z","shell.execute_reply.started":"2025-12-06T22:48:14.826951Z","shell.execute_reply":"2025-12-06T22:48:14.848403Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Mask processing","metadata":{}},{"cell_type":"code","source":"def _voxel_size_from_mm(size_mm, spacing_zyx):\n    \"\"\"\n    INPUT: \n    size_mm: desired mm size of the volume surrounding the lession\n    spacing_zyx: full volume array's spacing\n    \n    OUPTUT:\n    Dz, Dy, Dx = volume dimensions to take from the center of the zoomed in crop.\n    \"\"\"\n    Dz_mm, Dy_mm, Dx_mm = size_mm\n    sz, sy, sx = spacing_zyx\n    Dz = max(1, int(round(Dz_mm / sz)))\n    Dy = max(1, int(round(Dy_mm / sy)))\n    Dx = max(1, int(round(Dx_mm / sx)))\n    return Dz, Dy, Dx\n\ndef _find_slice_index_by_sop(slices, slice_id):\n    \"\"\"\n    INPUT: \n    slices: ordered list of dicom slices (no arrays, they ahve to be dicom files)\n    slice_id: id of the slice with lession, taken from the localizer's csv dataframe's SOPInstanceUID column\n\n    OUPTUT:\n    idx from the dicom slices list which has the lession or None if its a healthy patient and its not listed on localizers csv\n    \"\"\"\n    for idx, slc in enumerate(slices):\n        if getattr(slc, \"SOPInstanceUID\", None) == slice_id:\n            return idx\n    return None\n\ndef _crop_with_pad(vol, z0, y0, x0, Dz, Dy, Dx, pad_mode=\"edge\"):\n    \"\"\"\n    crop the image surrounding the lession and with the desired size. If the region will escape from the volume, pad by repeating the edges of the volume with data.\n    \n    INPUT: \n    vol: volume arrat\n    z0, x0, y0: volume's center coordinate.\n    Dz, Dy, Dx:  The added dimensions over the center to get the volume surrounding the lession.\n    pad_mode: for the np.pad function\n        \n    \n    OUPTUT:\n    idx from the dicom slices list which has the lession or None if its a healthy patient and its not listed on localizers csv\n    \n    Variables Cheat Sheet:\n     z1, y1, z1 = Volume to add in the three dimensions from the center z0, x0, y0\n    \"\"\"\n    Z,H,W = vol.shape\n    z1 = z0 + Dz\n    y1 = y0 + Dy\n    x1 = x0 + Dx\n\n    pb = (max(0,-z0), max(0,-y0), max(0,-x0))\n    pa = (max(0,z1-Z), max(0,y1-H), max(0,x1-W))\n    if any(pb) or any(pa):\n        vol = np.pad(vol, ((pb[0],pa[0]),(pb[1],pa[1]),(pb[2],pa[2])), mode=pad_mode)\n        z0 += pb[0]; y0 += pb[1]; x0 += pb[2]\n        z1 += pb[0]; y1 += pb[1]; x1 += pb[2]\n    return vol[z0:z1, y0:y1, x0:x1]\n\ndef _center_to_start(cz, cy, cx, Dz, Dy, Dx):\n    \"\"\"\n    Get the corner of the positive part in the mask in reference to the center of the image\n    INPUT: \n        cz, cy, cx: the original center of our zoomed in volume, AKA lesion location.\n        Dz, Dy, Dx: The dimension of the zoomed in volume with respect to the corner.\n    \n    OUPTUT:\n        cornerz, cornery, cornerx: zoom volume's coorinates\n    \"\"\"\n    cornerz = int(round(cz - Dz//2))\n    cornery = int(round(cy - Dy/2))\n    cornerx = int(round(cx - Dx/2))\n    return cornerz, cornery, cornerx","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.851023Z","iopub.execute_input":"2025-12-06T22:48:14.851432Z","iopub.status.idle":"2025-12-06T22:48:14.881382Z","shell.execute_reply.started":"2025-12-06T22:48:14.851399Z","shell.execute_reply":"2025-12-06T22:48:14.880017Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def lesion_mask_full_volume(\n    vol,\n    list_slices_raw,\n    spacing_zyx_raw,\n    spacing_zyx_iso,\n    lession_df,\n    series_uid,\n    size_mm=(90, 90, 90),\n):\n    \"\"\"\n    Build a full-volume binary mask based on lesion coordinates.\n    \n    INPUT\n    vol:            3D numpy array (Z_iso, H_iso, W_iso), already resampled to spacing_zyx_iso *1.2\n    list_slices_raw: list of original DICOM slices (raw space)\n    spacing_zyx_raw: (sz, sy, sx) in mm, from raw DICOM\n    spacing_zyx_iso: (tz, ty, tx) in mm, for the iso-resampled volume 'vol'\n    lession_df:     dataframe with columns [\"SeriesInstanceUID\",\"SOPInstanceUID\",\"x\",\"y\",...]\n    series_uid:     SeriesInstanceUID for this volume/patient\n    size_mm:        physical ROI size (Dz_mm, Dy_mm, Dx_mm) around lesion center, default as per top solution 4 from competition\n\n    Returns:\n        mask same shape as vol, 1 in ROI cube, 0 elsewhere\n        info (dict):  metadata for sanity checks - Used ONLY FOR DEBUGGING and personal references/testing\n    \"\"\"\n    #Get the shape of the volume, which will be resued\n    Z_iso, H_iso, W_iso = vol.shape\n\n    # raw and iso spacings\n    sz, sy, sx = map(float, spacing_zyx_raw)\n    tz, ty, tx = map(float, spacing_zyx_iso)\n    kz, ky, kx = (sz / tz, sy / ty, sx / tx)  # raw -> iso scaling factors\n\n    # raw dimensions from DICOM slices/instances\n    raw_H = int(getattr(list_slices_raw[0], \"Rows\", 0))\n    raw_W = int(getattr(list_slices_raw[0], \"Columns\", 0))\n    raw_Z = len(list_slices_raw)\n\n    # look for the patient in the lessions table/train_localisations.csv\n    rows = lession_df[lession_df[\"SeriesInstanceUID\"] == series_uid]\n    has_lesion = len(rows) > 0\n    cz = cy = cx = None\n\n    if has_lesion: #If patient has lession then we create a vinary volume if not its all zeroes\n        row = rows.iloc[0]\n        sop = str(row[\"SOPInstanceUID\"])\n        cz_raw = _find_slice_index_by_sop(list_slices_raw, sop)\n\n        if cz_raw is None:\n            has_lesion = False\n        else:\n            # raw x,y in pixel coordinates from localizer CSV\n            cx_raw = float(row[\"x\"])\n            cy_raw = float(row[\"y\"])\n\n            # clamp in raw grid\n            cx_raw = max(0, min(raw_W - 1, cx_raw))\n            cy_raw = max(0, min(raw_H - 1, cy_raw))\n\n            # map to iso grid\n            cz = int(round(cz_raw * kz))\n            cy = int(round(cy_raw * ky))\n            cx = int(round(cx_raw * kx))\n\n            # clamp in iso grid\n            cz = max(0, min(Z_iso - 1, cz))\n            cy = max(0, min(H_iso - 1, cy))\n            cx = max(0, min(W_iso - 1, cx))\n\n    # full-volume mask\n    mask = np.zeros_like(vol, dtype=np.uint8)\n\n    # no lesion -> all zeros\n    if not has_lesion:\n        info = {\n            \"label\": 0,\n            \"center_zyx_iso\": None,\n            \"start_zyx_iso\": None,\n            \"size_vox_iso\": None,\n            \"spacing_iso\": tuple(map(float, spacing_zyx_iso)),\n            \"series_uid\": str(series_uid),\n            \"raw_shape_guess\": (raw_Z, raw_H, raw_W),\n            \"scale_raw_to_iso\": (kz, ky, kx),\n        }\n        return mask, 0, info\n\n    # ROI size in voxels (in iso space)\n    Dz = max(1, int(round(size_mm[0] / tz)))\n    Dy = max(1, int(round(size_mm[1] / ty)))\n    Dx = max(1, int(round(size_mm[2] / tx)))\n\n    # corner in iso coords\n    z0, y0, x0 = _center_to_start(cz, cy, cx, Dz, Dy, Dx)\n\n    # clamp ROI so it stays inside volume #THIS CAN BE A VULNERABILITY and get the coordinates of the ROI/positive part of mask)\n    z_start = max(0, z0)\n    y_start = max(0, y0)\n    x_start = max(0, x0)\n    z_end   = min(Z_iso, z0 + Dz)\n    y_end   = min(H_iso, y0 + Dy)\n    x_end   = min(W_iso, x0 + Dx)\n\n    # set ROI region to 1\n    mask[z_start:z_end, y_start:y_end, x_start:x_end] = 1\n\n    info = {\n        \"label\": 1,\n        \"center_zyx_iso\": (int(cz), int(cy), int(cx)),\n        \"start_zyx_iso\": (int(z_start), int(y_start), int(x_start)),\n        \"size_vox_iso\": (int(Dz), int(Dy), int(Dx)),\n        \"spacing_iso\": tuple(map(float, spacing_zyx_iso)),\n        \"series_uid\": str(series_uid),\n        \"raw_shape_guess\": (raw_Z, raw_H, raw_W),\n        \"scale_raw_to_iso\": (kz, ky, kx),\n    }\n\n    return mask, 1, info #Last 2 for debugging and personal use\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T23:00:32.916401Z","iopub.execute_input":"2025-12-06T23:00:32.916728Z","iopub.status.idle":"2025-12-06T23:00:32.93524Z","shell.execute_reply.started":"2025-12-06T23:00:32.916706Z","shell.execute_reply":"2025-12-06T23:00:32.934106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def center_crop_or_pad(vol, target_shape, z_start=0.6, x_start=0.5, y_start=0.5):\n    \"\"\"\n    INPUT: \n    vol: Volume (full or mask)\n    target_shape: Target shgpe of the volume and ROI\n    start coordinate: id of the slice with lession, taken from the localizer's csv dataframe's SOPInstanceUID column\n    \n    OUPTUT:\n    idx from the dicom slices list which has the lession or None if its a healthy patient and its not listed on localizers csv\n    \"\"\"\n    vol_z, vol_y, vol_x = vol.shape\n    tz, ty, tx = target_shape\n    padz = max(tz - vol_z, 0)\n    pady = max(ty - vol_y, 0)\n    padx = max(tx - vol_x, 0)\n    vol = np.pad(vol,\n                 ((padz//2, padz - padz//2),\n                  (pady//2, pady - pady//2),\n                  (padx//2, padx - padx//2)),\n                 mode='constant')\n    z, y, x = vol.shape\n    cz = int(round(z_start * (z - 1)))\n    cy = int(round(y_start * (y - 1)))\n    cx = int(round(x_start * (x - 1)))\n    \n    startz, starty, startx = cz-tz//2, cy-ty//2, cx-tx//2\n    startz = max(0, min(z - tz, startz))\n    starty = max(0, min(y - ty, starty))\n    startx = max(0, min(x - tx, startx))\n                       \n    endz, endy, endx = startz + tz, starty + ty, startx + tx\n    return vol[startz:endz, starty:endy, startx:endx]\n\ndef mask_validity(original_mask, resampled_mask):\n    \"\"\"\n    mask: numpy array [D, H, W] after center_crop_or_pad\n    Returns: (is_valid: bool, reason: str) True if the mask was not lost False if the mask was completely lost\n    \"\"\"\n    # binarize in case it has been saved as float\n    #m = mask > eps\n    orig_sum = float(original_mask.sum())\n    new_sum  = float(resampled_mask.sum())\n    \n    if orig_sum > 0 and new_sum == 0:\n        print('mask lost')\n        return False\n\n    return True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T23:00:35.333106Z","iopub.execute_input":"2025-12-06T23:00:35.333429Z","iopub.status.idle":"2025-12-06T23:00:35.343744Z","shell.execute_reply.started":"2025-12-06T23:00:35.333409Z","shell.execute_reply":"2025-12-06T23:00:35.342529Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Loader","metadata":{}},{"cell_type":"code","source":"class Volume_mask_Dataset(Dataset):\n    def __init__(self, train_df=train_df, series_root=DICOM_PATH, \n                 target_iso_full=TARGET_ISO_BRAIN, \n                 roi_mm=(60,60,60), out_shape=TARGET_SIZE, loc_df=loc_df):\n        \n        self.df = train_df.reset_index(drop=True)\n        self.root = series_root\n        self.target_iso_full = target_iso_full\n        self.roi_mm = roi_mm\n        self.out_shape = out_shape\n        \n    def __len__(self):\n        return len(self.df)\n        \n    def __getitem__(self, idx):\n        sid = str(self.df[ID_COL].iloc[idx])\n        dicom_dir = os.path.join(self.root, sid)\n        \n        ############ Load the raw DICOM volume ##########################\n        modality, spacing_zyx_raw, sample_volume_raw, the_series_uid, dicom_slices = load_dicom_series(dicom_dir)\n\n\n        # Get the raw volume before resampling\n        sample_volume_raw = modality_specific_preprocessing(sample_volume_raw, modality, cta_startegy=\"hu_window\")\n        # Remove the leading frame dimension if it is 1, issue found while debugging\n        if sample_volume_raw.ndim == 4 and sample_volume_raw.shape[0] == 1:\n            sample_volume_raw = np.squeeze(sample_volume_raw, axis=0)\n\n        ############# Resample to isotropic spacing #######################\n        spacing_zyx_iso, volume_iso = resample_isotropic(sample_volume_raw, spacing_zyx_raw, target_iso=TARGET_ISO_BRAIN)\n        ############### Generate full-volume mask (1 in ROI, 0 elsewhere) ###############\n        original_mask, _, _ = lesion_mask_full_volume(volume_iso, \n                                                 dicom_slices, \n                                                 spacing_zyx_raw, \n                                                 spacing_zyx_iso, \n                                                 loc_df, \n                                                 series_uid=the_series_uid, \n                                                 size_mm=self.roi_mm)\n        resampled_mask = center_crop_or_pad(original_mask, TARGET_SIZE)\n        volume_iso = center_crop_or_pad(volume_iso, TARGET_SIZE)\n        return sid, volume_iso, original_mask, resampled_mask","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T23:00:45.978573Z","iopub.execute_input":"2025-12-06T23:00:45.978923Z","iopub.status.idle":"2025-12-06T23:00:45.988789Z","shell.execute_reply.started":"2025-12-06T23:00:45.978899Z","shell.execute_reply":"2025-12-06T23:00:45.987418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vol_dir = '/kaggle/working/volumes' #DIRECTORY TO SAVE PRE-PROCESSED VOLUMES\nmask_dir = '/kaggle/working/masks' #DIRECTORY TO SAVE PRE-PROCESSED VOLUMES' MASKS\n\n#EXCLUDES THOSE MASKS THAT GOT LOST WHEN STANDARDIZING VOLUMES TO SPECIFIC SIZE\ndef cache_dual_from_dataset(data: Volume_mask_Dataset, vol_dir=str, mask_dir=str):\n    '''\n    Apply pre-processing - standardize size - check if mask got lost - save volume and mask under same name but ddifferent folder.\n    \n    '''\n    os.makedirs(vol_dir, exist_ok=True)\n    os.makedirs(mask_dir, exist_ok=True)\n    succesful = []\n    unsuccesful = []\n\n    for t in tqdm(range(5)): #For testing\n    #for t in tqdm(range(len(data))): #For Running Full Data\n        try:\n            sid, x_full, original_mask, resampled_mask = ds[t]\n            is_valid = mask_validity(original_mask, resampled_mask)\n            if float(resampled_mask.sum()) > 0 and float(resampled_mask.sum()) >= 110592.0 * 0.8:\n                unsuccesful.append(sid)\n                print(f'Mask lost more than 20% of its volume standardizing vol size for: {sid}')\n            if not is_valid:\n                unsuccesful.append(sid)\n                print(f'Mask lost when standardizing vol size for: {sid}')\n                continue\n                \n            #print(is_valid)\n            out_vols = os.path.join(vol_dir, f\"{sid}.npz\")\n            out_mask = os.path.join(mask_dir, f\"{sid}.npz\")\n            \n            np.savez_compressed(out_vols, vol=x_full.astype(np.float16))\n            np.savez_compressed(out_mask, vol=resampled_mask.astype(np.float16))\n            succesful.append(sid)\n        except Exception as e:\n            print(e)\n            unsuccesful.append(sid)\n            \n    print(f\"Cached OK: {len(succesful)} | Failed: {len(unsuccesful)}\")\n    np.savez_compressed('/kaggle/working/bad.npz', lst=unsuccesful)\n    if unsuccesful[:5]:\n        print(\"Examples of failures:\", unsuccesful[:5])\n    return succesful, unsuccesful","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T23:03:48.295544Z","iopub.execute_input":"2025-12-06T23:03:48.295896Z","iopub.status.idle":"2025-12-06T23:03:48.30635Z","shell.execute_reply.started":"2025-12-06T23:03:48.295868Z","shell.execute_reply":"2025-12-06T23:03:48.305286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#CHANGE AS APPLICABLE\nquart =len(train_df)//4\ntrain_df_quart1 = train_df.iloc[:quart]\n# train_df_quart2 = train_df.iloc[quart:quart*2]\n# train_df_quart3 = train_df.iloc[quart*2:quart*3]\n# train_df_quart4 = train_df.iloc[quart*3:]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T23:03:50.470472Z","iopub.execute_input":"2025-12-06T23:03:50.470817Z","iopub.status.idle":"2025-12-06T23:03:50.476299Z","shell.execute_reply.started":"2025-12-06T23:03:50.470782Z","shell.execute_reply":"2025-12-06T23:03:50.47509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ds = Volume_mask_Dataset(train_df=train_df_quart1) #CHANGE as APPLICABLE\nok, bad = cache_dual_from_dataset(ds, vol_dir, mask_dir)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T23:03:51.234805Z","iopub.execute_input":"2025-12-06T23:03:51.235578Z","iopub.status.idle":"2025-12-06T23:04:24.262651Z","shell.execute_reply.started":"2025-12-06T23:03:51.235545Z","shell.execute_reply":"2025-12-06T23:04:24.261587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Paths - CHANGE AS APPLICABLE\nzip_path_masks = '/kaggle/working/masks_quart1.zip'\nzip_path_vols = '/kaggle/working/vols_quart2.zip'\n# Create zip\nshutil.make_archive(base_name=zip_path_vols.replace('.zip', ''), format='zip', root_dir=vol_dir)\nshutil.make_archive(base_name=zip_path_masks.replace('.zip', ''), format='zip', root_dir=mask_dir)\n\nprint(f\"Zipped to: {zip_path_masks}\")\nprint(f\"Zipped to: {zip_path_vols}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.937193Z","iopub.status.idle":"2025-12-06T22:48:14.937497Z","shell.execute_reply.started":"2025-12-06T22:48:14.93737Z","shell.execute_reply":"2025-12-06T22:48:14.937383Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Personal Use, Debugging Code","metadata":{}},{"cell_type":"markdown","source":"## Review Fails","metadata":{}},{"cell_type":"code","source":"# failed_patient= bad[1]\n# train_df_quart2.loc[train_df['SeriesInstanceUID'] == failed_patient]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.938735Z","iopub.status.idle":"2025-12-06T22:48:14.939133Z","shell.execute_reply.started":"2025-12-06T22:48:14.938912Z","shell.execute_reply":"2025-12-06T22:48:14.93895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# good_patient= ok[18]\n# train_df_quart2.loc[train_df['SeriesInstanceUID'] == good_patient]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.940215Z","iopub.status.idle":"2025-12-06T22:48:14.94062Z","shell.execute_reply.started":"2025-12-06T22:48:14.940424Z","shell.execute_reply":"2025-12-06T22:48:14.940442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# bad = np.load('/kaggle/working/bad.npz')[\"lst\"]\n# print(bad)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.944097Z","iopub.status.idle":"2025-12-06T22:48:14.944548Z","shell.execute_reply.started":"2025-12-06T22:48:14.944337Z","shell.execute_reply":"2025-12-06T22:48:14.944358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# roi_npz = np.load('/kaggle/working/masks/'+failed_patient +'.npz')[\"vol\"]\n# full_npz = np.load('/kaggle/working/volumes/'+failed_patient +'.npz')['vol']\n\n# print(full_npz.shape)\n# plt.imshow(full_npz[22])\n# plt.title(\"Slice of 3d Volume (DICOM series)\")\n# plt.show()\n\n# print(roi_npz.shape)\n# print(np.where(roi_npz.max(axis=(1,2)) == 1)[0])\n# print(len(np.where(roi_npz.max(axis=(1,2)) == 1)[0]))\n# print(roi_npz.shape)\n# plt.imshow(roi_npz[22])\n# plt.title(\"Slice of 3d MASK (DICOM series)\")\n# plt.show()\n\n# plt.imshow(roi_npz[22]*full_npz[22])\n# plt.title(\"Slice of 3d ROI (DICOM series)\")\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.946303Z","iopub.status.idle":"2025-12-06T22:48:14.947527Z","shell.execute_reply.started":"2025-12-06T22:48:14.947251Z","shell.execute_reply":"2025-12-06T22:48:14.947279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# roi_npz = np.load('/kaggle/working/masks/'+good_patient +'.npz')[\"vol\"]\n# full_npz = np.load('/kaggle/working/volumes/'+good_patient +'.npz')['vol']\n\n# print(full_npz.shape)\n# plt.imshow(full_npz[38])\n# plt.title(\"Slice of 3d Volume (DICOM series)\")\n# plt.show()\n\n# print(roi_npz.shape)\n# print(np.where(roi_npz.max(axis=(1,2)) == 1)[0])\n# print(len(np.where(roi_npz.max(axis=(1,2)) == 1)[0]))\n# print(roi_npz.shape)\n# plt.imshow(roi_npz[38])\n# plt.title(\"Slice of 3d MASK (DICOM series)\")\n# plt.show()\n\n# plt.imshow(roi_npz[38]*full_npz[38])\n# plt.title(\"Slice of 3d ROI (DICOM series)\")\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.949764Z","iopub.status.idle":"2025-12-06T22:48:14.950574Z","shell.execute_reply.started":"2025-12-06T22:48:14.950208Z","shell.execute_reply":"2025-12-06T22:48:14.950232Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Preview Files","metadata":{}},{"cell_type":"code","source":"# train_df_quart1.loc[9], train_df_quart1.loc[9]['SeriesInstanceUID']  #Check the data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.951727Z","iopub.status.idle":"2025-12-06T22:48:14.95217Z","shell.execute_reply.started":"2025-12-06T22:48:14.951917Z","shell.execute_reply":"2025-12-06T22:48:14.95196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# import glob\n# folder = \"/kaggle/working/masks\"\n# npz_files = sorted(glob.glob(os.path.join(folder, \"*.npz\")))\n\n# print(f\"Found {len(npz_files)} npz files\")\n# for i, f in enumerate(npz_files[:10]):  # show first 10 names\n#     print(i, \"->\", os.path.basename(f))\n\n# n = 9  # 0 = first file, 1 = second, etc.\n# file_path = npz_files[n]\n# print(\"Using file:\", file_path)\n\n# data = np.load(file_path)   # NpzFile object, behaves like a dict\n# print(\"Keys in this npz:\", data.files)\n\n# mask = data[data.files[0]]   # grab the first array\n# print(\"Image shape:\", mask.shape)\n# print(\"Image dtype:\", mask.dtype)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.953719Z","iopub.status.idle":"2025-12-06T22:48:14.954175Z","shell.execute_reply.started":"2025-12-06T22:48:14.953951Z","shell.execute_reply":"2025-12-06T22:48:14.953971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(mask.shape)\n# print(np.where(mask.max(axis=(1,2)) == 1)[0])\n# print(len(np.where(mask.max(axis=(1,2)) == 1)[0]))\n# plt.imshow(mask[80])\n# plt.title(\"Slice of 3d Volume (DICOM series)\")\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.954756Z","iopub.status.idle":"2025-12-06T22:48:14.955151Z","shell.execute_reply.started":"2025-12-06T22:48:14.954944Z","shell.execute_reply":"2025-12-06T22:48:14.954962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# folder = \"/kaggle/working/volumes\"\n# npz_files = sorted(glob.glob(os.path.join(folder, \"*.npz\")))\n\n# print(f\"Found {len(npz_files)} npz files\")\n# for i, f in enumerate(npz_files[:10]):  # show first 10 names\n#     print(i, \"->\", os.path.basename(f))\n\n# n = 9  # 0 = first file, 1 = second, etc.\n# file_path = npz_files[n]\n# print(\"Using file:\", file_path)\n\n# data = np.load(file_path)   # NpzFile object, behaves like a dict\n# print(\"Keys in this npz:\", data.files)\n\n# img = data[data.files[0]]   # grab the first array\n# print(\"Image shape:\", img.shape)\n# print(\"Image dtype:\", img.dtype)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.955707Z","iopub.status.idle":"2025-12-06T22:48:14.956074Z","shell.execute_reply.started":"2025-12-06T22:48:14.955869Z","shell.execute_reply":"2025-12-06T22:48:14.955883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(img.shape)\n# plt.imshow(img[92])\n# plt.title(\"Slice of 3d Volume (DICOM series)\")\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.957131Z","iopub.status.idle":"2025-12-06T22:48:14.957432Z","shell.execute_reply.started":"2025-12-06T22:48:14.957305Z","shell.execute_reply":"2025-12-06T22:48:14.957318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(img.shape)\n# plt.imshow(img[79]*mask[79])\n# plt.title(\"Slice of 3d Volume (DICOM series)\")\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-06T22:48:14.959874Z","iopub.status.idle":"2025-12-06T22:48:14.960331Z","shell.execute_reply.started":"2025-12-06T22:48:14.960098Z","shell.execute_reply":"2025-12-06T22:48:14.960132Z"}},"outputs":[],"execution_count":null}]}