{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":39272,"databundleVersionId":4629629},{"sourceType":"datasetVersion","sourceId":4619805,"datasetId":2688675,"databundleVersionId":4681402}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"0e0f6b04-24b4-406d-9f9a-0cc4b291d998","cell_type":"markdown","source":"# Breast Cancer Detection — Preprocessing (Paper Approach)\nFollowing: Prodan et al., *Applying Deep Learning Methods for Mammography Analysis and Breast Cancer Detection*, Applied Sciences 2023\n\n**Pipeline (Algorithm 1 from paper):**\n1. Load RSNA PNG images (theoviel dataset)\n2. Preprocessing: cropping + windowing + normalization\n3. Oversample positives 5:1\n4. Horizontal flip augmentation\n5. 5-fold stratified cross validation splits\n\n**Note:** StyleGAN-XL synthetic image generation is excluded (requires extended GPU training).\nAll other steps follow the paper exactly.","metadata":{}},{"id":"b9573604-3b90-4bb8-aba6-8055daf6ad0e","cell_type":"markdown","source":"## 1. Setup","metadata":{}},{"id":"3a322a8f-80c3-4f92-89cc-dad454140d7d","cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nfrom sklearn.model_selection import StratifiedKFold\nfrom tqdm import tqdm\nimport warnings\nfrom sklearn.utils import resample\n\nwarnings.filterwarnings('ignore')\n\nSEED        = 42\nTARGET_SIZE = (512, 512)   \nN_FOLDS     = 5           \nPOS_RATIO   = 2\n\nDATA_DIR = Path('/kaggle/input/datasets/theoviel/rsna-breast-cancer-512-pngs')\nCSV_PATH = Path('/kaggle/input/competitions/rsna-breast-cancer-detection/train.csv')\nOUT_DIR  = Path('/kaggle/working/processed')\n\nOUT_DIR.mkdir(parents=True, exist_ok=True)\n\nnp.random.seed(SEED)\nprint('Setup complete.')\nprint(f'Target size : {TARGET_SIZE}')\nprint(f'Folds       : {N_FOLDS}')\nprint(f'Pos ratio   : {POS_RATIO}:1')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T11:56:56.34418Z","iopub.execute_input":"2026-04-25T11:56:56.344577Z","iopub.status.idle":"2026-04-25T11:56:56.353931Z","shell.execute_reply.started":"2026-04-25T11:56:56.344548Z","shell.execute_reply":"2026-04-25T11:56:56.352979Z"}},"outputs":[],"execution_count":null},{"id":"360ce998-18f6-4444-8b28-3d76daa0a0f1","cell_type":"markdown","source":"## 2. Load Metadata","metadata":{}},{"id":"72777e31-1b67-4a47-9294-1a8edde1335a","cell_type":"code","source":"df = pd.read_csv(CSV_PATH)\n\n# Verify image paths exist using theoviel flat naming: {patient_id}_{image_id}.png\ndf['img_path'] = df.apply(\n    lambda r: str(DATA_DIR / f\"{r['patient_id']}_{r['image_id']}.png\"), axis=1\n)\n\n# Keep only rows whose images actually exist\ndf['exists'] = df['img_path'].apply(lambda p: Path(p).exists())\ndf = df[df['exists']].reset_index(drop=True)\n\nprint(f'Total images found : {len(df)}')\nprint(f'Positive (cancer=1): {df[\"cancer\"].sum()}')\nprint(f'Negative (cancer=0): {(df[\"cancer\"]==0).sum()}')\nprint(f'Imbalance ratio    : {(df[\"cancer\"]==0).sum() / df[\"cancer\"].sum():.1f}:1')\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T11:14:11.978069Z","iopub.execute_input":"2026-04-25T11:14:11.978482Z","iopub.status.idle":"2026-04-25T11:17:54.776663Z","shell.execute_reply.started":"2026-04-25T11:14:11.978454Z","shell.execute_reply":"2026-04-25T11:17:54.775727Z"}},"outputs":[],"execution_count":null},{"id":"5f90d2c9-7cd0-40df-9a1f-5d5e01663fb8","cell_type":"markdown","source":"## 3. Preprocessing Functions (Paper Section 3)\nThree steps as described in the paper: **cropping → windowing → normalization**","metadata":{}},{"id":"24cbde9c-4ad5-43fe-ba40-74e57ccc319b","cell_type":"code","source":"def crop(image):\n    \"\"\"\n    Step 1 — Cropping.\n    Remove black borders by finding the bounding box of non-zero pixels.\n    Equivalent to the 'cropping' step in the paper.\n    \"\"\"\n    if image is None:\n        return None\n    thresh = cv2.threshold(image, 5, 255, cv2.THRESH_BINARY)[1]\n    coords = cv2.findNonZero(thresh)\n    if coords is None:\n        return image\n    x, y, w, h = cv2.boundingRect(coords)\n    return image[y:y+h, x:x+w]\n\n\ndef windowing(image, low_pct=1, high_pct=99):\n    \"\"\"\n    Step 2 — Windowing.\n    Brightness adjustment that improves contrast between soft and dense tissue.\n    The paper describes this as: 'brightness adjustment gives much greater contrast\n    between soft and dense tissues, enhancing visualization of anatomical structures.'\n    Implementation: clip to percentile window then rescale to [0, 255].\n    \"\"\"\n    low  = np.percentile(image, low_pct)\n    high = np.percentile(image, high_pct)\n    image = np.clip(image, low, high)\n    image = ((image - low) / (high - low + 1e-8) * 255).astype(np.uint8)\n    return image\n\n\ndef normalize(image):\n    \"\"\"\n    Step 3 — Normalization.\n    Scale pixel values to [0, 1].\n    Final normalization with ImageNet stats happens inside the DataLoader.\n    \"\"\"\n    return (image.astype(np.float32) / 255.0 * 255).astype(np.uint8)\n\n\ndef preprocess(image_path, flip=False):\n    \"\"\"\n    Full pipeline: load -> crop -> windowing -> normalize -> resize.\n    Optional horizontal flip for augmentation (paper uses flip only).\n    Returns uint8 grayscale array — ImageNet normalization in DataLoader.\n    \"\"\"\n    img = cv2.imread(str(image_path), cv2.IMREAD_GRAYSCALE)\n    if img is None:\n        return None\n    img = crop(img)\n    if img is None:\n        return None\n    img = windowing(img)\n    img = normalize(img)\n    img = cv2.resize(img, TARGET_SIZE, interpolation=cv2.INTER_AREA)\n    if flip:\n        img = cv2.flip(img, 1)\n    return img\n\n\nprint('Preprocessing functions defined.')\nprint('Pipeline: crop -> windowing -> normalize -> resize -> [optional flip]')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T11:19:34.243554Z","iopub.execute_input":"2026-04-25T11:19:34.244407Z","iopub.status.idle":"2026-04-25T11:19:34.255413Z","shell.execute_reply.started":"2026-04-25T11:19:34.244368Z","shell.execute_reply":"2026-04-25T11:19:34.254598Z"}},"outputs":[],"execution_count":null},{"id":"c09e28b4-f8c3-476a-a7c0-0da71404bbb1","cell_type":"markdown","source":"## 4. Visualize Preprocessing Steps","metadata":{}},{"id":"d7de0fb1-eb45-4be2-b9cd-c25c875c10bc","cell_type":"code","source":"# Find a readable positive sample\nsample_row  = None\nsample_path = None\nfor _, row in df[df['cancer'] == 1].iterrows():\n    p   = Path(row['img_path'])\n    img = cv2.imread(str(p), cv2.IMREAD_GRAYSCALE)\n    if img is not None:\n        sample_row  = row\n        sample_path = p\n        raw         = img\n        break\n\ncropped   = crop(raw)\nwindowed  = windowing(cropped)\nnormed    = normalize(windowed)\nresized   = cv2.resize(normed, TARGET_SIZE, interpolation=cv2.INTER_AREA)\nflipped   = cv2.flip(resized, 1)\n\nsteps  = [raw, cropped, windowed, normed, resized, flipped]\ntitles = [\n    f'1. Original\\n{raw.shape}',\n    f'2. Crop\\n{cropped.shape}',\n    f'3. Windowing\\n{windowed.shape}',\n    f'4. Normalize\\n{normed.shape}',\n    f'5. Resize {TARGET_SIZE}\\n{resized.shape}',\n    f'6. H-Flip (aug)\\n{flipped.shape}'\n]\n\nfig, axes = plt.subplots(1, 6, figsize=(22, 4))\nfor ax, img, title in zip(axes, steps, titles):\n    ax.imshow(img, cmap='gray')\n    ax.set_title(title, fontsize=8)\n    ax.axis('off')\n\nplt.suptitle('Preprocessing Pipeline — Paper Algorithm 1 (Positive Sample)', fontweight='bold')\nplt.tight_layout()\nplt.savefig('/kaggle/working/preprocessing_steps.png', bbox_inches='tight')\nplt.show()\nprint(f'Sample: {sample_path.name}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T11:19:34.700721Z","iopub.execute_input":"2026-04-25T11:19:34.701496Z","iopub.status.idle":"2026-04-25T11:19:36.376426Z","shell.execute_reply.started":"2026-04-25T11:19:34.701466Z","shell.execute_reply":"2026-04-25T11:19:36.375448Z"}},"outputs":[],"execution_count":null},{"id":"6e744c58-1d76-4494-bfa6-8309ef607388","cell_type":"markdown","source":"## 5. Compare Windowing vs No Windowing\nDemonstrates why windowing improves tissue contrast as described in the paper.","metadata":{}},{"id":"adbb7117-b838-4b05-99da-f548a50b5a02","cell_type":"code","source":"fig, axes = plt.subplots(2, 4, figsize=(16, 8))\nwindow_params = [(1, 99), (2, 98), (5, 95), (10, 90)]\n\ncropped_sample = crop(raw)\n\nfor col, (lo, hi) in enumerate(window_params):\n    no_window  = normalize(cropped_sample)\n    with_window = windowing(cropped_sample, lo, hi)\n    axes[0, col].imshow(no_window,  cmap='gray'); axes[0, col].axis('off')\n    axes[0, col].set_title('No windowing', fontsize=8)\n    axes[1, col].imshow(with_window, cmap='gray'); axes[1, col].axis('off')\n    axes[1, col].set_title(f'Window [{lo}%, {hi}%]', fontsize=8)\n\nplt.suptitle('Effect of Windowing on Tissue Contrast', fontweight='bold')\nplt.tight_layout()\nplt.savefig('/kaggle/working/windowing_comparison.png', bbox_inches='tight')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T11:19:58.385867Z","iopub.execute_input":"2026-04-25T11:19:58.38678Z","iopub.status.idle":"2026-04-25T11:19:59.98119Z","shell.execute_reply.started":"2026-04-25T11:19:58.386739Z","shell.execute_reply":"2026-04-25T11:19:59.980086Z"}},"outputs":[],"execution_count":null},{"id":"09a04361-d966-4f54-9283-a3f515077273","cell_type":"markdown","source":"## 6. Process and Save All Images","metadata":{}},{"id":"f0bf58dd-9236-4df3-80bc-77ae87151b9c","cell_type":"code","source":"if (OUT_DIR / '0').exists() and len(list((OUT_DIR / '0').glob('*.png'))) > 100:\n    print('Already processed — loading from disk')\n    records = []\n    for label in [0, 1]:\n        for p in (OUT_DIR / str(label)).glob('*.png'):\n            # p.stem is the filename without extension\n            parts = p.stem.replace('_flip','').replace('_orig','').split('_')\n            records.append({\n                'filename': p.name, \n                'path': str(p),\n                'patient_id': parts[0], \n                'image_id': parts[1],\n                'cancer': label, \n                'augmented': 'flip' in p.stem\n            })\n    processed_df = pd.DataFrame(records)\n    print(f'Loaded {len(processed_df)} records from disk')\n\nelse:\n    # Initialize for fresh processing\n    for label in [0, 1]:\n        (OUT_DIR / str(label)).mkdir(parents=True, exist_ok=True)\n\n    records = []\n    skipped = 0\n\n    for _, row in tqdm(df.iterrows(), total=len(df), desc='Processing'):\n        label    = int(row['cancer'])\n        src_path = Path(row['img_path'])\n\n        # Original Image\n        img = preprocess(src_path, flip=False)\n        if img is None:\n            skipped += 1\n            continue\n\n        filename  = f\"{row['patient_id']}_{row['image_id']}_orig.png\"\n        save_path = OUT_DIR / str(label) / filename\n        cv2.imwrite(str(save_path), img)\n\n        # Helper to avoid repeating dict creation\n        base_record = {\n            'patient_id' : row['patient_id'],\n            'image_id'   : row['image_id'],\n            'cancer'     : label,\n            'age'        : row.get('age'),\n            'laterality' : row.get('laterality'),\n            'view'       : row.get('view'),\n            'density'    : row.get('density')\n        }\n\n        records.append({**base_record, 'filename': filename, 'path': str(save_path), 'augmented': False})\n\n        # Paper: horizontal flip for positives only (label == 1)\n        if label == 1:\n            img_flip      = preprocess(src_path, flip=True)\n            if img_flip is not None:\n                flip_filename = f\"{row['patient_id']}_{row['image_id']}_flip.png\"\n                flip_path     = OUT_DIR / str(label) / flip_filename\n                cv2.imwrite(str(flip_path), img_flip)\n                records.append({**base_record, 'filename': flip_filename, 'path': str(flip_path), 'augmented': True})\n\n    processed_df = pd.DataFrame(records)\n    print(f'Done. Processed: {len(processed_df)} | Skipped: {skipped}')\n    print(f'Positive images : {(processed_df[\"cancer\"]==1).sum()} (orig + flip)')\n    print(f'Negative images : {(processed_df[\"cancer\"]==0).sum()}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T11:20:11.402087Z","iopub.execute_input":"2026-04-25T11:20:11.402427Z","iopub.status.idle":"2026-04-25T11:43:41.594033Z","shell.execute_reply.started":"2026-04-25T11:20:11.402397Z","shell.execute_reply":"2026-04-25T11:43:41.592535Z"}},"outputs":[],"execution_count":null},{"id":"55db5e9b-fa22-4c1b-b5a5-a63f5ebb4a68","cell_type":"markdown","source":"## 7. Oversample Positives 5:1 (Paper Section 3)\nPaper states: *'we over-sample the positive examples in a ratio of 5:1'*","metadata":{}},{"id":"b777bf7f-98f7-4001-9bc3-0a7b44c197a4","cell_type":"code","source":"# Reduce negatives BEFORE oversampling — drop excess from processed_df\npos_df = processed_df[processed_df['cancer'] == 1]  # 2316\nneg_df = processed_df[processed_df['cancer'] == 0]  # 53548\n\n# Keep only enough negatives for 5:1 ratio against positives\nN_NEG_KEEP = len(pos_df) * POS_RATIO   # 2316 * 5 = 11580\n\nneg_df      = neg_df.sample(n=N_NEG_KEEP, random_state=SEED)\nprocessed_df = pd.concat([pos_df, neg_df]).sample(frac=1, random_state=SEED).reset_index(drop=True)\n\nprint(f'processed_df after negative reduction:')\nprint(f'  Positive : {(processed_df[\"cancer\"]==1).sum()}')\nprint(f'  Negative : {(processed_df[\"cancer\"]==0).sum()}')\nprint(f'  Total    : {len(processed_df)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T11:47:03.44098Z","iopub.execute_input":"2026-04-25T11:47:03.441571Z","iopub.status.idle":"2026-04-25T11:47:03.495651Z","shell.execute_reply.started":"2026-04-25T11:47:03.441521Z","shell.execute_reply":"2026-04-25T11:47:03.494637Z"}},"outputs":[],"execution_count":null},{"id":"a05aac32-69f6-4d8c-8d64-c8091d7c0a18","cell_type":"code","source":"\npos_df = processed_df[processed_df['cancer'] == 1]  # 2316\nneg_df = processed_df[processed_df['cancer'] == 0]  # 11580\n\n# Paper: oversample positives 5:1 → positives = negatives * 5\nN_POS_TARGET = len(neg_df) * POS_RATIO   # 11580 * 5 = 57900 — too many\n# Better interpretation: positives = 5x original positives\n# since we already reduced negatives to match\n\nN_POS_TARGET = len(neg_df)  # match negatives count first\n# Then upsample positives to be 5x negatives\npos_upsampled = resample(pos_df, replace=True, n_samples=len(neg_df) * POS_RATIO, random_state=SEED)\n\ndataset_df = pd.concat([neg_df, pos_upsampled]).sample(frac=1, random_state=SEED).reset_index(drop=True)\n\nprint(f'After 5:1 positive oversample:')\nprint(f'  Positive : {(dataset_df[\"cancer\"]==1).sum()}')\nprint(f'  Negative : {(dataset_df[\"cancer\"]==0).sum()}')\nprint(f'  Total    : {len(dataset_df)}')\nprint(f'  Ratio    : {(dataset_df[\"cancer\"]==1).sum() / (dataset_df[\"cancer\"]==0).sum():.1f}:1 (pos:neg)')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T11:57:14.182192Z","iopub.execute_input":"2026-04-25T11:57:14.183326Z","iopub.status.idle":"2026-04-25T11:57:14.236057Z","shell.execute_reply.started":"2026-04-25T11:57:14.183283Z","shell.execute_reply":"2026-04-25T11:57:14.234988Z"}},"outputs":[],"execution_count":null},{"id":"19f291dd-b6d3-49f8-91b3-3d80162ff7b8","cell_type":"markdown","source":"## 8. 5-Fold Stratified Cross Validation (Paper Section 3)\nPaper states: *'Stratified 5-fold Cross Validation so that each fold had approximately the same proportion of positive and negative samples'*","metadata":{}},{"id":"3d657ee7-e709-4fe3-bbb3-5cd05125aa00","cell_type":"code","source":"skf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=SEED)\n\ndataset_df['fold'] = -1\n\nfor fold_idx, (train_idx, val_idx) in enumerate(skf.split(dataset_df, dataset_df['cancer'])):\n    dataset_df.loc[val_idx, 'fold'] = fold_idx\n\nprint('Fold distribution:')\nfor fold in range(N_FOLDS):\n    fold_data = dataset_df[dataset_df['fold'] == fold]\n    n_pos     = (fold_data['cancer'] == 1).sum()\n    n_neg     = (fold_data['cancer'] == 0).sum()\n    print(f'  Fold {fold} : {len(fold_data):5d} samples | Pos: {n_pos:4d} | Neg: {n_neg:4d} | Ratio: {n_neg/n_pos:.1f}:1')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T11:57:17.13078Z","iopub.execute_input":"2026-04-25T11:57:17.131142Z","iopub.status.idle":"2026-04-25T11:57:17.167697Z","shell.execute_reply.started":"2026-04-25T11:57:17.131114Z","shell.execute_reply":"2026-04-25T11:57:17.166781Z"}},"outputs":[],"execution_count":null},{"id":"7ff27a3f-9a7c-49cc-8e9c-d2ee7d4450a4","cell_type":"markdown","source":"## 9. Save Dataset CSV with Fold Assignments","metadata":{}},{"id":"b2b0bab8-66bb-426f-8d9a-3fdb514ec151","cell_type":"code","source":"dataset_df.to_csv('/kaggle/working/dataset_with_folds.csv', index=False)\nprint(f'Saved: dataset_with_folds.csv ({len(dataset_df)} rows)')\nprint()\nprint('Usage in training:')\nprint('  for fold in range(5):')\nprint('      train_df = dataset_df[dataset_df[\"fold\"] != fold]')\nprint('      val_df   = dataset_df[dataset_df[\"fold\"] == fold]')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T11:57:20.902615Z","iopub.execute_input":"2026-04-25T11:57:20.902977Z","iopub.status.idle":"2026-04-25T11:57:21.127834Z","shell.execute_reply.started":"2026-04-25T11:57:20.902941Z","shell.execute_reply":"2026-04-25T11:57:21.12693Z"}},"outputs":[],"execution_count":null},{"id":"a7f15336-bab3-4155-8303-491dcd989faf","cell_type":"markdown","source":"## 10. Visual Sanity Check — Processed Samples","metadata":{}},{"id":"19ba59e5-d2f8-4729-80ca-937a3ecec563","cell_type":"code","source":"fig, axes = plt.subplots(2, 4, figsize=(14, 7))\n\nfor row_idx, label in enumerate([0, 1]):\n    samples   = dataset_df[dataset_df['cancer'] == label].head(4)\n    row_label = 'Negative' if label == 0 else 'Positive'\n    for col_idx, (_, s) in enumerate(samples.iterrows()):\n        img = cv2.imread(s['path'], cv2.IMREAD_GRAYSCALE)\n        if img is None:\n            axes[row_idx, col_idx].axis('off')\n            continue\n        aug_tag = '[flip]' if s['augmented'] else '[orig]'\n        axes[row_idx, col_idx].imshow(img, cmap='gray')\n        axes[row_idx, col_idx].set_title(f'{row_label} {aug_tag}\\nFold {s[\"fold\"]}', fontsize=8)\n        axes[row_idx, col_idx].axis('off')\n\nplt.suptitle('Processed Samples — Negative vs Positive', fontweight='bold')\nplt.tight_layout()\nplt.savefig('/kaggle/working/processed_samples.png', bbox_inches='tight')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T11:57:22.683392Z","iopub.execute_input":"2026-04-25T11:57:22.683735Z","iopub.status.idle":"2026-04-25T11:57:25.168537Z","shell.execute_reply.started":"2026-04-25T11:57:22.683706Z","shell.execute_reply":"2026-04-25T11:57:25.167737Z"}},"outputs":[],"execution_count":null},{"id":"df6e34ac-b13a-4369-aa7d-2e84a68950be","cell_type":"markdown","source":"## 11. Class Distribution Plot","metadata":{}},{"id":"72ae79da-6f89-4c4e-8f4d-134ec6a25d8f","cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(12, 4))\n\n# Overall distribution\ncounts = dataset_df['cancer'].value_counts().sort_index()\naxes[0].bar(['Negative', 'Positive'], counts.values,\n            color=['steelblue', 'tomato'], edgecolor='black', linewidth=0.7)\naxes[0].set_title('Overall Class Distribution (5:1 ratio)')\naxes[0].set_ylabel('Count')\nfor i, v in enumerate(counts.values):\n    axes[0].text(i, v + 20, str(v), ha='center', fontsize=10)\n\n# Per fold distribution\nfold_pos = [dataset_df[dataset_df['fold']==f]['cancer'].sum() for f in range(N_FOLDS)]\nfold_neg = [(dataset_df['fold']==f).sum() - p for f, p in enumerate(fold_pos)]\nx        = np.arange(N_FOLDS)\nw        = 0.35\naxes[1].bar(x - w/2, fold_neg, w, label='Negative', color='steelblue', edgecolor='black', linewidth=0.5)\naxes[1].bar(x + w/2, fold_pos, w, label='Positive', color='tomato',    edgecolor='black', linewidth=0.5)\naxes[1].set_xticks(x)\naxes[1].set_xticklabels([f'Fold {i}' for i in range(N_FOLDS)])\naxes[1].set_title('Distribution per Fold')\naxes[1].set_ylabel('Count')\naxes[1].legend()\n\nplt.suptitle('5-Fold Stratified Cross Validation — Class Balance', fontweight='bold')\nplt.tight_layout()\nplt.savefig('/kaggle/working/fold_distribution.png', bbox_inches='tight')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T11:57:25.170574Z","iopub.execute_input":"2026-04-25T11:57:25.171296Z","iopub.status.idle":"2026-04-25T11:57:25.976229Z","shell.execute_reply.started":"2026-04-25T11:57:25.171248Z","shell.execute_reply":"2026-04-25T11:57:25.975322Z"}},"outputs":[],"execution_count":null},{"id":"3e1d479c-4caa-4b79-9e8f-0a80a910485e","cell_type":"markdown","source":"## 12. Summary","metadata":{}},{"id":"927ac309-a0f4-4595-acb7-f9fe6b54f706","cell_type":"code","source":"n_pos_total = (dataset_df['cancer'] == 1).sum()\nn_neg_total = (dataset_df['cancer'] == 0).sum()\n\nprint('=== Preprocessing Summary (Paper Approach) ===')\nprint(f'Reference         : Prodan et al., Applied Sciences 2023')\nprint(f'Target resolution : {TARGET_SIZE[0]}x{TARGET_SIZE[1]} px')\nprint()\nprint('Pipeline:')\nprint('  Step 1 — Crop      : remove black borders (bounding box of non-zero pixels)')\nprint('  Step 2 — Windowing : percentile brightness adjustment [1%, 99%] for tissue contrast')\nprint('  Step 3 — Normalize : scale to [0, 255] uint8')\nprint('  Step 4 — Resize    : to 512x512')\nprint('  Step 5 — Flip      : horizontal flip on positive cases only')\nprint()\nprint('Dataset:')\nprint(f'  Total images      : {len(dataset_df)}')\nprint(f'  Positive          : {n_pos_total} (orig + flip)')\nprint(f'  Negative          : {n_neg_total} (sampled for 5:1 ratio)')\nprint(f'  Actual ratio      : {n_neg_total/n_pos_total:.2f}:1')\nprint()\nprint('Cross Validation:')\nprint(f'  Strategy          : {N_FOLDS}-fold stratified')\nprint(f'  Each fold         : ~{len(dataset_df)//N_FOLDS} samples, same pos/neg ratio')\nprint()\nprint('Not included (requires extended training):')\nprint('  StyleGAN-XL synthetic image generation')\nprint('  (paper used 1000 synthetic positives — pretrained weights at zenodo.org/record/7794274)')\nprint()\nprint('Output files in /kaggle/working/:')\nprint('  dataset_with_folds.csv   <- main file for training')\nprint('  processed/0/*.png        <- negative images')\nprint('  processed/1/*.png        <- positive images (orig + flip)')\nprint('  preprocessing_steps.png')\nprint('  windowing_comparison.png')\nprint('  processed_samples.png')\nprint('  fold_distribution.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T11:57:29.747159Z","iopub.execute_input":"2026-04-25T11:57:29.747551Z","iopub.status.idle":"2026-04-25T11:57:29.759711Z","shell.execute_reply.started":"2026-04-25T11:57:29.747522Z","shell.execute_reply":"2026-04-25T11:57:29.758561Z"}},"outputs":[],"execution_count":null}]}