{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":13451,"datasetId":654585,"databundleVersionId":1188070}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-22T07:51:26.342263Z","iopub.execute_input":"2025-07-22T07:51:26.342855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\n\nshutil.make_archive(\"/kaggle/working/brain_window_images_resnet\", 'zip', \"/kaggle/working/brain_window_images_resnet\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls -lh /kaggle/working/\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"1+1","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fileFolder = '/kaggle/working/brain_window_images_anyNone/'","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls -lh /kaggle/working/brain_window_images_anyNone.zip","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls -lh /kaggle/working/brain_window_images_resnet.zip\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from google.colab import drive\ndrive.mount('/content/drive')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!find /kaggle/working/brain_window_images_resnet -type f | wc -l","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!find /kaggle/working/brain_window_images_resnet -type f -exec du -b {} + | \\\n  awk -F/ '{ext=$NF; sub(\".*\\\\.\", \"\", ext); size[ext]+=$1; count[ext]++} END {for (e in size) printf \"%s\\t%d files\\t%.2f MB\\n\", e, count[e], size[e]/(1024*1024)}' | sort -k3 -hr\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!curl --upload-file /kaggle/working/brain_window_images_resnet.zip https://transfer.sh/brain_window_images_resnet.zip","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from google.colab import files\nuploaded = files.upload()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.applications.resnet50 import ResNet50, preprocess_input\n\nfrom tensorflow.keras.preprocessing import image\nfrom tqdm import tqdm\nimport zipfile\nimport shutil\n\n# --------------------\n# Paths\n#input_dir = '/kaggle/input/your_input_folder'  # Change this!\ninput_dir = fileFolder\noutput_dir = '/kaggle/working/brain_window_images_resnet'\nzip_path = '/kaggle/working/brain_window_images_resnet.zip'\nos.makedirs(output_dir, exist_ok=True)\n\n# --------------------\n# Load Pretrained ResNet50\nmodel = ResNet50(weights='imagenet', include_top=False, pooling='avg')  # Use GAP features\n\n# --------------------\n# Helper: Extract Features\ndef extract_features(img_path):\n    img = image.load_img(img_path, target_size=(224, 224))\n    x = image.img_to_array(img)\n    x = np.expand_dims(x, axis=0)\n    x = preprocess_input(x)\n    features = model.predict(x, verbose=0)\n    return features.squeeze()\n\n# --------------------\n# Iterate & Save Features\nfor fname in tqdm(sorted(os.listdir(input_dir))):\n    if fname.lower().endswith(('.png', '.jpg', '.jpeg', '.bmp')):\n        fpath = os.path.join(input_dir, fname)\n        features = extract_features(fpath)\n        np.save(os.path.join(output_dir, fname + '.npy'), features)\n\n# --------------------\n# Zip the output_dir\nshutil.make_archive(output_dir, 'zip', output_dir)\n\nprint(f\"✅ Done! Features saved in: {output_dir}\")\nprint(f\"📦 Zipped version at: {zip_path}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls -l /kaggle/working/","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'/kaggle/working/brain_window_images_resnet.zip'\nfrom IPython.display import FileLink\n\nFileLink('/kaggle/working/brain_window_images_resnet.zip')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import HTML\n\nHTML('<a href=\"/kaggle/working/brain_window_images_resnet.zip\" download>Click here to download</a>')\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from kaggle_notebook_utils import UserSessionClient\nUserSessionClient().download('/kaggle/working/brain_window_images_resnet.zip')\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!mv /kaggle/working/brain_window_images_resnet.zip ./brain_window_images_resnet.zip\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Start","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n#import cudf as pd\nimport pydicom\nimport os\nimport matplotlib.pyplot as plt\nimport collections\nfrom tqdm import tqdm_notebook as tqdm\nfrom datetime import datetime\n\nfrom math import ceil, floor, log\nimport cv2\n\nimport tensorflow as tf\nimport keras\n\nimport sys\n\n# from keras_applications.resnet import ResNet50\nfrom keras_applications.inception_v3 import InceptionV3\n\nfrom sklearn.model_selection import ShuffleSplit\n\n\n# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nfrom glob import glob\nfrom random import shuffle\nimport cv2\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\n#from keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom keras.layers import Convolution1D, concatenate, SpatialDropout1D, GlobalMaxPool1D, GlobalAvgPool1D, Embedding, \\\n    Conv2D, SeparableConv1D, Add, BatchNormalization, Activation, GlobalAveragePooling2D, LeakyReLU, Flatten\nfrom keras.layers import Dense, Input, Dropout, MaxPooling2D, Concatenate, GlobalMaxPooling2D, GlobalAveragePooling2D, \\\n    Lambda, Multiply, LSTM, Bidirectional, PReLU, MaxPooling1D\n#from keras.layers.pooling import _GlobalPooling1D\nfrom tensorflow.keras.layers import GlobalAveragePooling1D, GlobalMaxPooling1D\n\n#from keras.losses import mae, sparse_categorical_crossentropy, binary_crossentropy\nfrom tensorflow.keras.losses import MeanAbsoluteError, SparseCategoricalCrossentropy, BinaryCrossentropy\n\n\nfrom keras.models import Model\nfrom keras.applications.nasnet import NASNetMobile, NASNetLarge, preprocess_input\nfrom keras.optimizers import Adam, RMSprop\nfrom keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom imgaug import augmenters as iaa\nimport imgaug as ia\n#print(os.listdir(\"../input/sevenfeildmodel\"))\n\n# Any results you write to the current directory are saved as output.","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!pip install keras-applications\n#!pip install imgaug\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\ntest_images_dir = '../input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_test/'\ntrain_images_dir = '../input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train/'","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# IF GPU\nimport cupy as cp\ndef sigmoid_window_gpu(dcm, img, window_center, window_width, U=1.0, eps=(1.0 / 255.0)):\n    img = cp.array(np.array(img))\n    _, _, intercept, slope = get_windowing(dcm)\n    img = img * slope + intercept\n    ue = cp.log((U / eps) - 1.0)\n    W = (2 / window_width) * ue\n    b = ((-2 * window_center) / window_width) * ue\n    z = W * img + b\n    img = U / (1 + cp.power(np.e, -1.0 * z))\n    img = (img - cp.min(img)) / (cp.max(img) - cp.min(img))\n    return cp.asnumpy(img)\n\ndef correct_dcm(dcm):\n    #print('needs Correction')\n    x = dcm.pixel_array + 1000\n    px_mode = 4096\n    x[x>=px_mode] = x[x>=px_mode] - px_mode\n    dcm.PixelData = x.tobytes()\n    dcm.RescaleIntercept = -1000\n\n\ndef sigmoid_window(img, window_center, window_width, U=1.0, eps=(1.0 / 255.0),gpu=False):\n    if gpu:\n        return sigmoid_window_gpu(img, img.pixel_array, window_center, window_width, U, eps)\n    else:\n        _, _, intercept, slope = get_windowing(img)\n        img = img.pixel_array * slope + intercept\n        ue = np.log((U / eps) - 1.0)\n        W = (2 / window_width) * ue\n        b = ((-2 * window_center) / window_width) * ue\n        z = W * img + b\n        img = U / (1 + np.power(np.e, -1.0 * z))\n        img = (img - np.min(img)) / (np.max(img) - np.min(img))\n        return img\n\n\ndef sigmoid_brain_window(img,gpu=False):\n    return sigmoid_window(img, 40, 80,gpu=gpu)\n\ndef sigmoid_bsb_window(img,gpu=False):\n\n    if (img.BitsStored == 12) and (img.PixelRepresentation == 0) and (int(img.RescaleIntercept) > -100):\n        print('needCoreection')\n        correct_dcm(img)\n\n    brain_img = sigmoid_window(img, 40, 80,gpu=gpu)\n    subdural_img = sigmoid_window(img, 80, 200,gpu=gpu)\n    bone_img = sigmoid_window(img, 600, 2000,gpu=gpu)\n\n    bsb_img = np.zeros((brain_img.shape[0], brain_img.shape[1], 3))\n    bsb_img[:, :, 0] = brain_img\n    bsb_img[:, :, 1] = subdural_img\n    bsb_img[:, :, 2] = bone_img\n    return bsb_img\n\ndef get_first_of_dicom_field_as_int(x):\n    #get x[0] as in int is x is a 'pydicom.multival.MultiValue', otherwise get int(x)\n    if type(x) == pydicom.multival.MultiValue:\n        return int(x[0])\n    else:\n        return int(x)\n\n\ndef get_windowing(data):\n\n    dicom_fields = [data[('0028','1050')].value, #window center\n                    data[('0028','1051')].value, #window width\n                    data[('0028','1052')].value, #intercept\n                    data[('0028','1053')].value] #slope\n    return [get_first_of_dicom_field_as_int(x) for x in dicom_fields]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, cv2, pydicom\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom glob import glob\nfrom tqdm import tqdm\nimport tensorflow as tf\nfrom sklearn.model_selection import GroupShuffleSplit\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_images_dir","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dicom = pydicom.dcmread(train_images_dir + \"ID_036db39b7\" + \".dcm\")\n#dicom = pydicom.dcmread(train_images_dir + \"ID_000039fa0\" + \".dcm\")\n#\"\"\nimage=sigmoid_bsb_window(dicom)\n#image=sigmoid_bsb_window(dicom,gpu=True)\nplt.imshow(image, cmap=plt.cm.bone)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Brain Window Save","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport pydicom\nimport numpy as np\nfrom tqdm import tqdm\n\n# Apply Brain Window: center=40, width=80\ndef apply_brain_window(image, center=40, width=80):\n    min_val = center - width // 2\n    max_val = center + width // 2\n    windowed = np.clip(image, min_val, max_val)\n    windowed = ((windowed - min_val) / (max_val - min_val)) * 255.0\n    return windowed.astype(np.uint8)\n\n# Convert DICOM to PNG with Brain Window\ndef process_brain_window_images(input_dir, output_dir, limit=None):\n    os.makedirs(output_dir, exist_ok=True)\n    dcm_files = sorted([f for f in os.listdir(input_dir) if f.lower().endswith(\".dcm\")])\n\n    if limit:\n        dcm_files = dcm_files[:limit]\n\n    for filename in tqdm(dcm_files, desc=\"Applying Brain Window\"):\n        try:\n            dicom_path = os.path.join(input_dir, filename)\n            dicom = pydicom.dcmread(dicom_path)\n            \n            if not hasattr(dicom, 'pixel_array'):\n                print(f\"Warning: {filename} has no pixel data.\")\n                continue\n            \n            image = dicom.pixel_array.astype(np.float32)\n            windowed_img = apply_brain_window(image)\n\n            output_path = os.path.join(output_dir, filename.replace(\".dcm\", \".png\"))\n            cv2.imwrite(output_path, windowed_img)\n        except Exception as e:\n            print(f\"Error processing {filename}: {e}\")\n\n# Example usage (set your actual paths)\ninput_dicom_dir = \"/kaggle/input/rsna-intracranial-hemorrhage-detection/stage_1_train_images\"\noutput_image_dir = \"/kaggle/working/brain_window_images\"\nprocess_brain_window_images(input_dicom_dir, output_image_dir, limit=5000)  # You can remove 'limit' for full run","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Correct path from your earlier message\ninput_dicom_dir = \"../input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train/\"\noutput_image_dir = \"/kaggle/working/brain_window_images\"\n\nprocess_brain_window_images(input_dicom_dir, output_image_dir, limit=5000)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ninput_dicom_dir = \"../input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train/\"\ndcm_files = [f for f in os.listdir(input_dicom_dir) if f.lower().endswith(\".dcm\")]\n\nprint(f\"Total DICOM files in input folder: {len(dcm_files)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"752803/5000","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"csv_path = \"/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train.csv\"\ndf = pd.read_csv(csv_path)\ndf.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[df.Label==1]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load the dataset\ncsv_path = \"/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train.csv\"\ndf = pd.read_csv(csv_path)\n\n# Split 'ID' column into Image ID and Label Type\ndf[['prefix', 'Image_ID', 'Label_Type']] = df['ID'].str.extract(r'(^ID_)?([a-f0-9]+)_(\\w+)')\n\n# Drop the prefix column as it's not needed\ndf.drop(columns=['ID', 'prefix'], inplace=True)\n\n# Pivot to wide format (label types become columns)\ndf_wide = df.pivot_table(index='Image_ID', columns='Label_Type', values='Label', aggfunc='first').reset_index()\n\n# Optional: Fill NaNs (if any) with 0\ndf_wide.fillna(0, inplace=True)\n\n# (Optional) Create Patient ID (if different in dataset, for now same as Image ID)\ndf_wide['Patient_ID'] = df_wide['Image_ID']  # If patient info available, replace accordingly\n\n# Final columns: Patient_ID, Image_ID, labels...\ncols = ['Patient_ID', 'Image_ID'] + sorted([c for c in df_wide.columns if c not in ['Patient_ID', 'Image_ID']])\ndf_final = df_wide[cols]\n\n# Show\nprint(df_final.head())\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_final.sum()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport cv2\nfrom tqdm import tqdm\n\n# ✅ Step 1: Balance the data based on 'any' label in df_final\ndf_pos = df_final[df_final['any'] == 1].sample(n=107933, random_state=42)\ndf_neg = df_final[df_final['any'] == 0].sample(n=107933, random_state=42)\ndf_balanced = pd.concat([df_pos, df_neg]).reset_index(drop=True)\n\n# 🔘 Optional: Save balanced metadata\ndf_balanced.to_csv(\"/kaggle/working/brain_any_balanced_meta.csv\", index=False)\n\n# ✅ Step 2: Define Brain Windowing function\ndef brain_window(dcm):\n    try:\n        img = dcm.pixel_array * dcm.RescaleSlope + dcm.RescaleIntercept\n    except:\n        img = dcm.pixel_array\n    img = np.clip(img, 40, 80)\n    img = ((img - 40) / 40 * 255.0).astype(np.uint8)\n    return img\n\n# ✅ Step 3: Process & save brain-windowed images\ndef save_brain_window_images(image_ids, input_dir, output_dir):\n    os.makedirs(output_dir, exist_ok=True)\n    for img_id in tqdm(image_ids, desc=\"Processing BrainWindow Images\"):\n        dcm_path = os.path.join(input_dir, f\"{img_id}.dcm\")\n        if not os.path.exists(dcm_path):\n            continue\n        try:\n            dcm = pydicom.dcmread(dcm_path)\n            img = brain_window(dcm)\n            cv2.imwrite(os.path.join(output_dir, f\"{img_id}.png\"), img)\n        except Exception as e:\n            print(f\"[ERROR] {img_id}: {e}\")\n\n# ✅ Step 4: Run processing on the balanced set\ndicom_dir = \"/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train\"\noutput_dir = \"/kaggle/working/brain_window_images_anyNone\"\nsave_brain_window_images(df_balanced[\"Image_ID\"].tolist(), dicom_dir, output_dir)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nfrom tensorflow.keras.applications.resnet50 import ResNet50, preprocess_input\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.models import Model\nimport tensorflow as tf\nfrom tqdm import tqdm\nimport zipfile\n\n# Paths\n#input_folder = '/kaggle/working/output_dir'\ninput_folder = \"/kaggle/working/brain_window_images_anyNone\"\noutput_folder = '/kaggle/working/brain_window_images_resnet'\nzip_path = '/kaggle/working/brain_window_images_resnet.zip'\n\nos.makedirs(output_folder, exist_ok=True)\n\n# Load ResNet50 model (feature extractor)\nbase_model = ResNet50(weights='imagenet', include_top=False, pooling='avg')\nmodel = Model(inputs=base_model.input, outputs=base_model.output)\n\n# Process and save features\nfor fname in tqdm(sorted(os.listdir(input_folder))):\n    if not fname.lower().endswith(('.png', '.jpg', '.jpeg')):\n        continue\n    try:\n        img_path = os.path.join(input_folder, fname)\n        img = image.load_img(img_path, target_size=(224, 224))\n        x = image.img_to_array(img)\n        x = np.expand_dims(x, axis=0)\n        x = preprocess_input(x)\n        features = model.predict(x, verbose=0)\n        features = np.squeeze(features)\n        np.save(os.path.join(output_folder, fname + '.npy'), features)\n    except Exception as e:\n        print(f\"Error processing {fname}: {e}\")\n\n# Zip the output folder\nwith zipfile.ZipFile(zip_path, 'w') as zipf:\n    for root, _, files in os.walk(output_folder):\n        for file in files:\n            file_path = os.path.join(root, file)\n            arcname = os.path.relpath(file_path, output_folder)\n            zipf.write(file_path, arcname)\n\nprint(f\"\\n✅ Done! Download zip: {zip_path}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, cv2, pydicom\nimport numpy as np\nfrom tqdm import tqdm\n\ndef brain_window(dcm):\n    intercept = dcm.RescaleIntercept if \"RescaleIntercept\" in dcm else 0\n    slope = dcm.RescaleSlope if \"RescaleSlope\" in dcm else 1\n    img = dcm.pixel_array * slope + intercept\n    img = np.clip(img, 40, 80)  # Brain window\n    img = ((img - 40) / 40 * 255.0).astype(np.uint8)\n    return img\n\ndef save_brain_window_images(image_ids, input_dir, output_dir):\n    os.makedirs(output_dir, exist_ok=True)\n    saved, missing = 0, 0\n    for img_id in tqdm(image_ids):\n        dcm_path = os.path.join(input_dir, f\"ID_{img_id}.dcm\")\n        if not os.path.exists(dcm_path):\n            missing += 1\n            continue\n        try:\n            dcm = pydicom.dcmread(dcm_path)\n            img = brain_window(dcm)\n            cv2.imwrite(os.path.join(output_dir, f\"{img_id}.png\"), img)\n            saved += 1\n        except Exception as e:\n            print(f\"Error processing {img_id}: {e}\")\n    print(f\"Saved: {saved} images | Missing: {missing} files\")\n\n# Set paths and run\ndicom_dir = \"/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train\"\noutput_dir = \"/kaggle/working/brain_window_images_anyNone\"\nsave_brain_window_images(df_balanced[\"Image_ID\"].tolist(), dicom_dir, output_dir)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#output_dir\n#!ls '/kaggle/working/brain_window_images_anyNone/'\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#df_balanced[\"Image_ID\"].tolist()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ls ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load the CSV\ncsv_path = \"/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train.csv\"\ndf = pd.read_csv(csv_path)\n\n# Extract Image ID and Label Type\ndf['Image_ID'] = df['ID'].apply(lambda x: '_'.join(x.split('_')[:2]))\ndf['Label'] = df['ID'].apply(lambda x: x.split('_')[2])\n\n# Pivot into one-hot label columns (one row per Image_ID)\none_hot_df = df.pivot(index='Image_ID', columns='Label', values='LabelName').fillna(0)\none_hot_df = df.pivot(index='Image_ID', columns='Label', values='Label').fillna(0)\none_hot_df = df.pivot(index='Image_ID', columns='Label', values='Label').fillna(0)\n\n# Merge Label values into the one-hot format (0/1 values)\none_hot_df = df.pivot(index='Image_ID', columns='Label', values='Label').fillna(0)\n\n# Make column names consistent and reset index\none_hot_df.columns.name = None\none_hot_df.reset_index(inplace=True)\n\n# Add Patient_ID by trimming Image_ID (first 12 chars)\none_hot_df['Patient_ID'] = one_hot_df['Image_ID'].str[:12]\n\n# Reorder columns: Patient_ID, Image_ID, one-hot labels\ncols = ['Patient_ID', 'Image_ID'] + sorted([col for col in one_hot_df.columns if col not in ['Patient_ID', 'Image_ID']])\nfinal_df = one_hot_df[cols]\n\n# Preview\nprint(final_df.head())\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}