{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"accelerator":"GPU","colab":{"gpuType":"T4","provenance":[]},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":39272,"databundleVersionId":4629629,"isSourceIdPinned":false}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **EDA — Breast Cancer Mammography**","metadata":{"id":"P6k8PnlO2MuI"}},{"cell_type":"code","source":"import os\nimport glob\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport cv2\n\nPATH_DATASET = \"/kaggle/input/competitions/rsna-breast-cancer-detection\"","metadata":{"id":"bENt-uc92MuL","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:16:15.785971Z","iopub.execute_input":"2026-05-10T19:16:15.786898Z","iopub.status.idle":"2026-05-10T19:16:15.792358Z","shell.execute_reply.started":"2026-05-10T19:16:15.786849Z","shell.execute_reply":"2026-05-10T19:16:15.791467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv(os.path.join(PATH_DATASET, \"train.csv\"))\ndisplay(df_train.head())\nprint(f'Total scans: {len(df_train)}')","metadata":{"id":"7J6BSc6y2MuL","outputId":"fa4576ed-94a2-409b-b934-3ea398fb7e95","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:16:15.798033Z","iopub.execute_input":"2026-05-10T19:16:15.798346Z","iopub.status.idle":"2026-05-10T19:16:15.907585Z","shell.execute_reply.started":"2026-05-10T19:16:15.798324Z","shell.execute_reply":"2026-05-10T19:16:15.906806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ax = df_train.groupby(\"patient_id\").size().hist(bins=10)\nax.set_xlabel(\"Number of scans per patient\")\nax.set_ylabel(\"Number of patients\")\nplt.title(\"Scans Distribution\")\nplt.show()","metadata":{"id":"dtNDyoZi2MuM","outputId":"6513841e-c10a-4a34-9a7b-ffa34113d035","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:16:15.908875Z","iopub.execute_input":"2026-05-10T19:16:15.909077Z","iopub.status.idle":"2026-05-10T19:16:16.161957Z","shell.execute_reply.started":"2026-05-10T19:16:15.909057Z","shell.execute_reply":"2026-05-10T19:16:16.161414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cols = [\"site_id\", \"laterality\", \"view\", \"implant\", \"density\", \"machine_id\",\n        \"cancer\", \"biopsy\", \"invasive\", \"BIRADS\", \"difficult_negative_case\"]\n\nnb_rows = int(np.ceil(len(cols) / 2))\nfig, axarr = plt.subplots(ncols=2, nrows=nb_rows, figsize=(12, 5 * nb_rows))\n\nfor i, col in enumerate(cols):\n    df_train[col].value_counts().plot.pie(ax=axarr[i // 2, i % 2], autopct=\"%.1f%%\")\n    axarr[i // 2, i % 2].set_title(f\"Distribution of {col}\")\n\nplt.tight_layout()\nplt.show()","metadata":{"id":"lMA1bk6M2MuM","outputId":"ddd4949b-62c0-41fc-cb62-1847ea6baaaa","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:16:16.162876Z","iopub.execute_input":"2026-05-10T19:16:16.163178Z","iopub.status.idle":"2026-05-10T19:16:17.131422Z","shell.execute_reply.started":"2026-05-10T19:16:16.163144Z","shell.execute_reply":"2026-05-10T19:16:17.130595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train_patients = df_train.groupby(\"patient_id\").max()\nplt.figure(figsize=(10, 6))\ndf_train_patients.groupby(\"cancer\")[\"age\"].hist(bins=45, alpha=0.5, legend=True)\nplt.title(\"Age Distribution vs Cancer Status\")\nplt.xlabel(\"Age\")\nplt.show()","metadata":{"id":"ysSJsDEb2MuN","outputId":"470212e4-d855-4607-f750-b561bf9abb86","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:16:17.133017Z","iopub.execute_input":"2026-05-10T19:16:17.133348Z","iopub.status.idle":"2026-05-10T19:16:18.993461Z","shell.execute_reply.started":"2026-05-10T19:16:17.133324Z","shell.execute_reply":"2026-05-10T19:16:18.992702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ════════════════════════════════════════════════════════════════\n# CELL 6 — Incremental Cache Setup  ✅ v13\n# ✅ بيحسب عدد الـ chunks تلقائياً بناءً على صور السرطان المتاحة\n# ✅ بيوزع صور السرطان بالتساوي على كل chunk\n# ✅ مفيش Oversampling — cancer ratio حقيقية\n# ════════════════════════════════════════════════════════════════\n!pip install -q SimpleITK pydicom\n\nimport gc\nimport os\nimport threading\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport SimpleITK as sitk\nimport tensorflow as tf\nfrom concurrent.futures import ThreadPoolExecutor\n\n# ══════════════════════════════════════════════\n#  حساب تلقائي بناءً على الداتا الحقيقية\n# ══════════════════════════════════════════════\nCHUNK_SIZE      = 5_000\nN_VAL_PER_CHUNK = 500\nDCM_DIR         = os.path.join(PATH_DATASET, \"train_images\")\n\n# عدد صور السرطان الحقيقي في الداتا\n_df_c_all = df_train[df_train[\"cancer\"] == 1]\n_df_h_all = df_train[df_train[\"cancer\"] == 0]\n_n_cancer_total  = len(_df_c_all)\n_n_healthy_total = len(_df_h_all)\n\n# كام chunk ممكن؟ — كل chunk يحتاج n_c_per_chunk cancer\n# n_c_per_chunk = ما هو متاح من cancer مقسوم على عدد الـ chunks\n# نختار عدد chunks أولاً بناءً على TOTAL_SAMPLES\n_n_chunks_target = TOTAL_SAMPLES // CHUNK_SIZE\n\n# السرطان المتاح للـ train (بعد إزالة val)\n# n_val_c × n_chunks = cancer في الـ val\n# نحل المعادلة: n_c_per_chunk = (_n_cancer_total - n_val_c × n_chunks) / n_chunks\n# نجرب عدد chunks من الأكبر للأصغر\n_best_n_chunks = 1\nfor nc in range(_n_chunks_target, 0, -1):\n    _n_val_c_try   = round(N_VAL_PER_CHUNK * (_n_cancer_total / CHUNK_SIZE))\n    _n_val_c_total = _n_val_c_try * nc\n    _n_train_c_avail = _n_cancer_total - _n_val_c_total\n    if _n_train_c_avail >= nc:   # على الأقل 1 cancer per chunk في الـ train\n        _best_n_chunks = nc\n        break\n\n# الآن نحسب التوزيع الحقيقي\nN_CHUNKS        = _best_n_chunks\n# cancer per chunk = كل السرطان مقسوم على N_CHUNKS\n_N_C_PER_CHUNK  = _n_cancer_total  // N_CHUNKS\n_N_VAL_C        = round(N_VAL_PER_CHUNK * _N_C_PER_CHUNK / CHUNK_SIZE)\n_N_TRAIN_C      = _N_C_PER_CHUNK - _N_VAL_C\n# healthy = الباقي عشان يكمّل CHUNK_SIZE\n_N_H_PER_CHUNK  = CHUNK_SIZE - _N_C_PER_CHUNK\n_N_VAL_H        = N_VAL_PER_CHUNK - _N_VAL_C\n_N_TRAIN_H      = _N_H_PER_CHUNK - _N_VAL_H\nCANCER_RATIO    = _N_C_PER_CHUNK / CHUNK_SIZE\n\nprint(f\"✅ Dataset Stats:\")\nprint(f\"   Cancer  total : {_n_cancer_total:,}\")\nprint(f\"   Healthy total : {_n_healthy_total:,}\")\nprint(f\"\")\nprint(f\"✅ Auto-calculated Config v13:\")\nprint(f\"   N_CHUNKS       : {N_CHUNKS}\")\nprint(f\"   CHUNK_SIZE     : {CHUNK_SIZE:,}\")\nprint(f\"   TOTAL_SAMPLES  : {N_CHUNKS * CHUNK_SIZE:,}\")\nprint(f\"   Cancer/chunk   : {_N_C_PER_CHUNK:,}  ({CANCER_RATIO:.1%})\")\nprint(f\"   Healthy/chunk  : {_N_H_PER_CHUNK:,}  ({1-CANCER_RATIO:.1%})\")\nprint(f\"   Val cancer     : {_N_VAL_C:,} per chunk\")\nprint(f\"   Val healthy    : {_N_VAL_H:,} per chunk\")\nprint(f\"   Train cancer   : {_N_TRAIN_C:,} per chunk\")\nprint(f\"   Train healthy  : {_N_TRAIN_H:,} per chunk\")\n\n\n# ══════════════════════════════════════════════\n#  CumulativeValTracker\n# ══════════════════════════════════════════════\nclass CumulativeValTracker:\n    def __init__(self):\n        self.all_labels    = []\n        self.all_probs     = []\n        self.chunk_metrics = []\n\n    def add_chunk_results(self, labels, probs, chunk_id):\n        self.all_labels.append(labels)\n        self.all_probs.append(probs)\n        preds        = (probs >= 0.5).astype(int)\n        acc          = (preds == labels).mean()\n        cancer_mask  = labels == 1\n        healthy_mask = labels == 0\n        sensitivity  = preds[cancer_mask].mean()      if cancer_mask.sum()  > 0 else 0.0\n        specificity  = (1-preds[healthy_mask]).mean() if healthy_mask.sum() > 0 else 0.0\n        self.chunk_metrics.append({\n            \"chunk_id\"   : chunk_id,\n            \"n_samples\"  : len(labels),\n            \"n_cancer\"   : int(cancer_mask.sum()),\n            \"n_healthy\"  : int(healthy_mask.sum()),\n            \"acc\"        : acc,\n            \"sensitivity\": sensitivity,\n            \"specificity\": specificity,\n        })\n        return acc\n\n    def cumulative_accuracy(self):\n        if not self.all_labels: return 0.0\n        all_l = np.concatenate(self.all_labels)\n        all_p = np.concatenate(self.all_probs)\n        return ((all_p >= 0.5).astype(int) == all_l).mean()\n\n    def cumulative_auc(self):\n        from sklearn.metrics import roc_auc_score\n        if not self.all_labels: return 0.0\n        all_l = np.concatenate(self.all_labels)\n        all_p = np.concatenate(self.all_probs)\n        return roc_auc_score(all_l, all_p) if len(np.unique(all_l)) > 1 else 0.0\n\n    def print_summary(self, chunk_id):\n        cumul_acc = self.cumulative_accuracy()\n        try:    auc_str = f\"{self.cumulative_auc():.4f}\"\n        except: auc_str = \"N/A\"\n        total_n = sum(m[\"n_samples\"] for m in self.chunk_metrics)\n        print(f\"\\n{'='*70}\")\n        print(f\"  📊 Cumulative Val Summary — After Chunk #{chunk_id}\")\n        print(f\"{'='*70}\")\n        print(f\"  Total val samples seen  : {total_n:,}\")\n        print(f\"  ✅ Cumulative Accuracy   : {cumul_acc:.4f}  ({cumul_acc*100:.2f}%)\")\n        print(f\"  ✅ Cumulative AUC        : {auc_str}\")\n        print(f\"\\n  Per-chunk breakdown:\")\n        print(f\"  {'Chunk':<8} {'n':>6} {'cancer':>8} {'healthy':>8} {'ratio':>7} {'acc':>7} {'sens':>7} {'spec':>7}\")\n        print(f\"  {'-'*65}\")\n        for m in self.chunk_metrics:\n            ratio = m[\"n_cancer\"] / max(m[\"n_samples\"], 1)\n            print(f\"  #{m['chunk_id']:02d}    \"\n                  f\"{m['n_samples']:>6,} \"\n                  f\"{m['n_cancer']:>8,} \"\n                  f\"{m['n_healthy']:>8,} \"\n                  f\"{ratio:>7.1%} \"\n                  f\"{m['acc']:>7.4f} \"\n                  f\"{m['sensitivity']:>7.4f} \"\n                  f\"{m['specificity']:>7.4f}\")\n        print(f\"{'='*70}\\n\")\n\n\n# ══════════════════════════════════════════════\n#  IncrementalCacheLoader  ✅ v13\n# ══════════════════════════════════════════════\nclass IncrementalCacheLoader:\n    \"\"\"\n    ✅ v13 — بيحسب التوزيع تلقائياً من صور السرطان المتاحة\n    ✅ مفيش Oversampling — cancer ratio حقيقية بالضبط\n    ✅ كل chunk فيه نفس عدد السرطان بالضبط\n    \"\"\"\n\n    def __init__(self, df_full,\n                 n_chunks    = N_CHUNKS,\n                 chunk_size  = CHUNK_SIZE,\n                 n_val       = N_VAL_PER_CHUNK,\n                 dcm_dir     = DCM_DIR,\n                 img_size    = IMG_SIZE):\n\n        self.chunk_size = chunk_size\n        self.n_val      = n_val\n        self.dcm_dir    = dcm_dir\n        self.img_size   = img_size\n\n        # ── فصل السرطان عن الـ Healthy وخلط ─────────────────\n        df_c = df_full[df_full[\"cancer\"] == 1].copy().sample(\n            frac=1, random_state=42).reset_index(drop=True)\n        df_h = df_full[df_full[\"cancer\"] == 0].copy().sample(\n            frac=1, random_state=42).reset_index(drop=True)\n\n        # ── حساب التوزيع بناءً على المتاح ───────────────────\n        n_c_per_chunk = len(df_c) // n_chunks        # توزيع متساوي للسرطان\n        n_val_c       = round(n_val * n_c_per_chunk / chunk_size)\n        n_val_h       = n_val - n_val_c\n        n_h_per_chunk = chunk_size - n_c_per_chunk   # healthy يكمّل الـ chunk\n        n_train_c     = n_c_per_chunk - n_val_c\n        n_train_h     = n_h_per_chunk - n_val_h\n\n        # تحقق إن الـ healthy كافي\n        if len(df_h) < n_h_per_chunk * n_chunks:\n            n_chunks = len(df_h) // n_h_per_chunk\n            print(f\"⚠️ Healthy limited → adjusted to {n_chunks} chunks\")\n\n        self.n_chunks      = n_chunks\n        self.n_c_per_chunk = n_c_per_chunk\n        self.n_h_per_chunk = n_h_per_chunk\n        self.n_val_c       = n_val_c\n        self.n_val_h       = n_val_h\n        self.n_train_c     = n_train_c\n        self.n_train_h     = n_train_h\n        self.cancer_ratio  = n_c_per_chunk / chunk_size\n\n        # ── Val pool منفصل ───────────────────────────────────\n        n_val_c_total = n_val_c * n_chunks\n        n_val_h_total = n_val_h * n_chunks\n\n        df_c_val   = df_c.iloc[:n_val_c_total].copy()\n        df_c_train = df_c.iloc[n_val_c_total:n_val_c_total + n_train_c * n_chunks].copy().reset_index(drop=True)\n\n        df_h_val   = df_h.iloc[:n_val_h_total].copy()\n        df_h_train = df_h.iloc[n_val_h_total:n_val_h_total + n_train_h * n_chunks].copy().reset_index(drop=True)\n\n        # ── بناء chunks ──────────────────────────────────────\n        self.train_chunks = []\n        self.val_chunks   = []\n\n        for i in range(n_chunks):\n            c_tr = df_c_train.iloc[i * n_train_c : (i+1) * n_train_c].copy()\n            h_tr = df_h_train.iloc[i * n_train_h : (i+1) * n_train_h].copy()\n            train_chunk = pd.concat([c_tr, h_tr]).sample(\n                frac=1, random_state=i*10).reset_index(drop=True)\n\n            c_vl = df_c_val.iloc[i * n_val_c : (i+1) * n_val_c].copy()\n            h_vl = df_h_val.iloc[i * n_val_h : (i+1) * n_val_h].copy()\n            val_chunk = pd.concat([c_vl, h_vl]).sample(\n                frac=1, random_state=i*10+1).reset_index(drop=True)\n\n            self.train_chunks.append(train_chunk)\n            self.val_chunks.append(val_chunk)\n\n        self.current_chunk_idx = -1\n        self.train_images = None\n        self.train_labels = None\n        self.val_images   = None\n        self.val_labels   = None\n\n        self._clahe      = cv2.createCLAHE(clipLimit=3.0, tileGridSize=(8, 8))\n        self._clahe_lock = threading.Lock()\n\n        # ── ملخص ─────────────────────────────────────────────\n        print(f\"\\n✅ IncrementalCacheLoader v13\")\n        print(f\"   Chunks          : {n_chunks} × {chunk_size:,} images\")\n        print(f\"   Cancer/chunk    : {n_c_per_chunk:,}  ({n_c_per_chunk/chunk_size:.1%})\")\n        print(f\"   Healthy/chunk   : {n_h_per_chunk:,}  ({n_h_per_chunk/chunk_size:.1%})\")\n        print(f\"   Val/chunk       : {n_val_c:,}c + {n_val_h:,}h = {n_val:,}\")\n        print(f\"   Train/chunk     : {n_train_c:,}c + {n_train_h:,}h = {n_train_c+n_train_h:,}\")\n        print()\n\n        # ── Verify ───────────────────────────────────────────\n        ok = True\n        for i, (tr, vl) in enumerate(zip(self.train_chunks, self.val_chunks)):\n            nc_tr = (tr[\"cancer\"]==1).sum()\n            nh_tr = (tr[\"cancer\"]==0).sum()\n            nc_vl = (vl[\"cancer\"]==1).sum()\n            nh_vl = (vl[\"cancer\"]==0).sum()\n            match = (nc_tr == n_train_c and nh_tr == n_train_h and\n                     nc_vl == n_val_c   and nh_vl == n_val_h)\n            status = \"✅\" if match else \"⚠️\"\n            print(f\"   {status} Chunk #{i+1}: train=({nc_tr}c+{nh_tr}h) | val=({nc_vl}c+{nh_vl}h)\")\n            if not match: ok = False\n        if ok:\n            print(f\"\\n   🎯 All {n_chunks} chunks perfectly balanced!\")\n        print()\n\n    @property\n    def num_chunks(self):\n        return len(self.train_chunks)\n\n    def _read_one(self, args):\n        path_info, lat = args\n        try:\n            p   = os.path.join(self.dcm_dir,\n                               str(path_info[0]),\n                               str(path_info[1]) + \".dcm\")\n            img = sitk.ReadImage(p)\n            arr = sitk.GetArrayFromImage(img).squeeze().astype(np.float32)\n\n            if (img.HasMetaDataKey(\"0028|0004\") and\n                    \"MONOCHROME1\" in img.GetMetaData(\"0028|0004\")):\n                arr = arr.max() - arr\n\n            arr = (arr - arr.min()) / (arr.max() - arr.min() + 1e-6)\n            arr = (arr * 255).astype(np.uint8)\n\n            mask = arr > 0\n            r, c = mask.any(1), mask.any(0)\n            if r.any() and c.any():\n                arr = arr[np.ix_(r, c)]\n\n            with self._clahe_lock:\n                arr = self._clahe.apply(arr)\n\n            if lat == \"R\":\n                arr = cv2.flip(arr, 1)\n\n            arr = cv2.resize(arr, (self.img_size, self.img_size))\n            arr = cv2.cvtColor(arr, cv2.COLOR_GRAY2RGB)\n            return arr, True\n\n        except Exception:\n            return np.zeros((self.img_size, self.img_size, 3), dtype=np.uint8), False\n\n    def _load_df(self, df):\n        args = list(zip(\n            df[[\"patient_id\", \"image_id\"]].values,\n            df[\"laterality\"].values\n        ))\n        results = []\n        for arg in args:                          # loop بدل ThreadPool لتوفير RAM\n            results.append(self._read_one(arg))\n\n        imgs_list, ok_list = zip(*results)\n        imgs   = np.array(imgs_list, dtype=np.uint8)\n        labels = df[\"cancer\"].values.astype(np.int32)\n        ok     = np.array(ok_list, dtype=bool)\n\n        return imgs[ok], labels[ok]\n\n    def next_chunk(self):\n        nxt = self.current_chunk_idx + 1\n        if nxt >= len(self.train_chunks):\n            print(\"✅ All chunks completed!\")\n            return None\n\n        self.train_images = self.train_labels = None\n        self.val_images   = self.val_labels   = None\n        gc.collect()\n        try:\n            tf.keras.backend.clear_session()\n        except Exception:\n            pass\n\n        self.current_chunk_idx = nxt\n        train_df = self.train_chunks[nxt]\n        val_df   = self.val_chunks[nxt]\n\n        print(f\"\\n🔄 Loading Chunk #{nxt+1}/{len(self.train_chunks)}\")\n        print(f\"   CSV → train: {(train_df['cancer']==1).sum():,}c + {(train_df['cancer']==0).sum():,}h\")\n        print(f\"   CSV → val  : {(val_df['cancer']==1).sum():,}c + {(val_df['cancer']==0).sum():,}h\")\n\n        print(\"   Loading train images...\")\n        tr_imgs, tr_lbls = self._load_df(train_df)\n\n        print(\"   Loading val images...\")\n        vl_imgs, vl_lbls = self._load_df(val_df)\n\n        self.train_images = tr_imgs\n        self.train_labels = tr_lbls\n        self.val_images   = vl_imgs\n        self.val_labels   = vl_lbls\n\n        nc_tr = (tr_lbls == 1).sum()\n        nc_vl = (vl_lbls == 1).sum()\n        ram_mb = (tr_imgs.nbytes + vl_imgs.nbytes) / 1e6\n\n        print(f\"   ✅ Train loaded: {len(tr_lbls):,} → {nc_tr:,}c ({nc_tr/max(len(tr_lbls),1):.1%}) + {(tr_lbls==0).sum():,}h\")\n        print(f\"   ✅ Val   loaded: {len(vl_lbls):,} → {nc_vl:,}c ({nc_vl/max(len(vl_lbls),1):.1%}) + {(vl_lbls==0).sum():,}h\")\n        print(f\"   RAM ≈ {ram_mb:.0f} MB\")\n\n        return nxt + 1\n\n\n# ── تهيئة ─────────────────────────────────────────────────────\ncache_loader = IncrementalCacheLoader(df_train)\nval_tracker  = CumulativeValTracker()\nprint(f\"\\n✅ Ready — {cache_loader.num_chunks} chunks × {CHUNK_SIZE:,} images\")\nprint(f\"   Cancer ratio : {cache_loader.cancer_ratio:.1%} (real, no oversampling)\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:17:57.694994Z","iopub.execute_input":"2026-05-10T19:17:57.695417Z","iopub.status.idle":"2026-05-10T19:18:00.992869Z","shell.execute_reply.started":"2026-05-10T19:17:57.695383Z","shell.execute_reply":"2026-05-10T19:18:00.992165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train[\"prediction_id\"] = df_train.apply(lambda r: f\"{r['patient_id']}_{r['laterality']}\", axis=1)\ndf_train_stat = df_train.groupby(\"prediction_id\").max()\n\nstat = df_train_stat.groupby(\"laterality\")[\"cancer\"].mean().to_dict()\nprint(f\"Stats by Laterality: {stat}\")","metadata":{"id":"vPTP6ZwD2MuN","outputId":"d5daff06-c4b6-4987-8976-f575ed2a1049","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:18:11.55756Z","iopub.execute_input":"2026-05-10T19:18:11.557856Z","iopub.status.idle":"2026-05-10T19:18:15.065886Z","shell.execute_reply.started":"2026-05-10T19:18:11.557825Z","shell.execute_reply":"2026-05-10T19:18:15.065038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test = pd.read_csv(os.path.join(PATH_DATASET, \"test.csv\"))\npreds = []\n\nfor _, row in df_test.iterrows():\n    preds.append({\n        \"prediction_id\": row[\"prediction_id\"],\n        \"cancer\": stat.get(row[\"laterality\"], 0.02)\n    })\n\ndf_preds = pd.DataFrame(preds).groupby('prediction_id').max()\ndf_preds.to_csv(\"submission.csv\")\n\n!head submission.csv","metadata":{"id":"2M8laHEr2MuO","outputId":"e6a98b06-f5d3-4b55-d8b1-473c6c08c547","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:18:23.572282Z","iopub.execute_input":"2026-05-10T19:18:23.572708Z","iopub.status.idle":"2026-05-10T19:18:23.726005Z","shell.execute_reply.started":"2026-05-10T19:18:23.572664Z","shell.execute_reply":"2026-05-10T19:18:23.725029Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **End EDA**\n\n---\n\n# **Preprocessing**","metadata":{"id":"0eyV6bBW2MuO"}},{"cell_type":"code","source":"# Cell 10 — Preprocessing helpers\n!pip install -q pylibjpeg pylibjpeg-libjpeg gdcm\nimport SimpleITK as sitk\nimport cv2, numpy as np\n\nclahe_global = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n\ndef preprocess_image_cv2(img_path: str, laterality: str) -> np.ndarray:\n    try:\n        img_sitk = sitk.ReadImage(img_path)\n        img = sitk.GetArrayFromImage(img_sitk).squeeze().astype(np.float32)\n\n        # MONOCHROME1 invert\n        if img_sitk.HasMetaDataKey('0028|0004') and \\\n           'MONOCHROME1' in img_sitk.GetMetaData('0028|0004'):\n            img = img.max() - img\n\n        # Normalize → uint8\n        img = ((img - img.min()) / (img.max() - img.min() + 1e-6) * 255).astype(np.uint8)\n\n    except Exception as e:\n        print(f\"Read error: {e}\")\n        return None\n\n    # Auto-crop\n    mask = img > 0\n    rows, cols = mask.any(1), mask.any(0)\n    if not rows.any() or not cols.any():\n        return None\n    img = img[np.ix_(rows, cols)]\n\n    # CLAHE\n    img = clahe_global.apply(img)\n\n    # Flip if Right\n    if laterality == 'R':\n        img = cv2.flip(img, 1)\n\n    # Grayscale → RGB\n    img = cv2.cvtColor(img, cv2.COLOR_GRAY2RGB)\n    return img\n\nprint('✅ preprocess_image_cv2 ready! (SimpleITK — handles all DCM compression)')","metadata":{"id":"WsiNMxdw2MuO","outputId":"1d748287-c2a6-4a20-df2e-a411ecc93126","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:18:28.489366Z","iopub.execute_input":"2026-05-10T19:18:28.489998Z","iopub.status.idle":"2026-05-10T19:18:33.660149Z","shell.execute_reply.started":"2026-05-10T19:18:28.489963Z","shell.execute_reply":"2026-05-10T19:18:33.65939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import SimpleITK as sitk\n\n# Test Preprocessing on a random image\nif df_train.empty:\n    print('❌ df_train is empty!')\nelse:\n    sample_row  = df_train.iloc[0]\n    sample_path = os.path.join(PATH_DATASET, 'train_images',\n                               str(sample_row['patient_id']),\n                               str(sample_row['image_id']) + '.dcm')\n\n    print(f\"Path: {sample_path}\")\n    print(f\"Exists: {os.path.exists(sample_path)}\")\n\n    processed = preprocess_image_cv2(sample_path, sample_row['laterality'])\n\n    if processed is not None:\n        fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(10, 5))\n\n        # ✅ Read raw with SimpleITK — handles all DCM compression\n        img_sitk = sitk.ReadImage(sample_path)\n        raw = sitk.GetArrayFromImage(img_sitk).squeeze().astype(np.float32)\n        raw = ((raw - raw.min()) / (raw.max() - raw.min() + 1e-6) * 255).astype(np.uint8)\n\n        ax1.imshow(raw, cmap='gray')\n        ax1.set_title('Before Preprocessing')\n        ax1.axis('off')\n\n        ax2.imshow(processed[:,:,0], cmap='gray')\n        ax2.set_title('After CLAHE + Crop')\n        ax2.axis('off')\n\n        plt.tight_layout()\n        plt.show()\n        print(f'Output shape: {processed.shape}')\n    else:\n        print('❌ Image not found or corrupted')","metadata":{"id":"wRmubL9Z2MuP","outputId":"80392d96-23cc-4e31-92b5-c390107e9e90","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:18:39.170836Z","iopub.execute_input":"2026-05-10T19:18:39.171152Z","iopub.status.idle":"2026-05-10T19:18:51.482221Z","shell.execute_reply.started":"2026-05-10T19:18:39.17112Z","shell.execute_reply":"2026-05-10T19:18:51.481375Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n\n# **TensorFlow Model — EfficientNetB2 (Fast + Accurate)**\n\n> ### التحسينات المضافة لتحقيق السرعة والدقة:\n> | التحسين | التأثير |\n> |---|---|\n> | **EfficientNetB2** بدل B0 | دقة أعلى بفارق واضح |\n> | **Mixed Precision fp16** | الـ GPU أسرع بـ ~2x |\n> | **8000 صورة فقط** (stratified) | تدريب أسرع بكثير |\n> | **Cache في RAM** | القراءة من الـ Drive مرة واحدة فقط |\n> | **Batch=32** | استغلال أفضل للـ GPU |\n> | **Oversampling قوي 25%** | الموديل يتعلم السرطان أحسن |\n> | **Focal Loss** | أفضل للـ imbalanced data |\n> | **Threshold Tuning** | كشف السرطان بدقة أعلى |\n> | **Cosine LR Decay** | تدريب أستقر وأسرع convergence |","metadata":{"id":"3ctVnvdS2MuP"}},{"cell_type":"code","source":"# ════════════════════════════════════════════════════════════════\n# CELL 1 — Install & Imports\n# ════════════════════════════════════════════════════════════════\n!pip install -q tensorflow scikit-learn seaborn pydicom SimpleITK\n\nimport os, cv2, numpy as np, pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pydicom\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers, callbacks, mixed_precision\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import (\n    classification_report, confusion_matrix,\n    roc_auc_score, roc_curve, precision_recall_curve\n)\n\n# ✅ Mixed Precision — fp16 على GPU = ~2x سرعة\nmixed_precision.set_global_policy('mixed_float16')\n\nprint(f'TensorFlow   : {tf.__version__}')\nprint(f'Compute dtype: {mixed_precision.global_policy().compute_dtype}')\ngpus = tf.config.list_physical_devices('GPU')\nprint(f'GPUs detected: {len(gpus)}')\nfor g in gpus:\n    tf.config.experimental.set_memory_growth(g, True)\n","metadata":{"id":"Ruu1H9412MuP","outputId":"4b0a97f6-d129-4a12-fbaa-676463b3f1ee","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:21:32.4911Z","iopub.execute_input":"2026-05-10T19:21:32.491577Z","iopub.status.idle":"2026-05-10T19:21:37.935591Z","shell.execute_reply.started":"2026-05-10T19:21:32.491545Z","shell.execute_reply":"2026-05-10T19:21:37.934624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 2 — Config ✅ v13\nSAVE_DIR  = \"/kaggle/working/model_checkpoints\"\nIMG_SIZE  = 380\nBATCH     = 16\nEPOCHS    = 60\n\n# ✅ Incremental settings\nTOTAL_SAMPLES        = 40_000\nCHUNK_SIZE           = 5_000\nN_SAMPLES            = TOTAL_SAMPLES\nN_VAL_PER_CHUNK      = 500\n# ✅ v13: TARGET_CANCER_RATIO بيتحسب تلقائياً في Cell 6\n# بس محتاجين نحط قيمة مبدئية عشان الـ print ميكسرش\nTARGET_CANCER_RATIO  = \"auto\"\nLABEL_SMOOTHING      = 0.05\nWARMUP_EPOCHS        = 3\n\nos.makedirs(SAVE_DIR, exist_ok=True)\nprint(f\"✅ Config v13 ready!\")\nprint(f\"   IMG_SIZE            : {IMG_SIZE}\")\nprint(f\"   TOTAL_SAMPLES       : {TOTAL_SAMPLES:,}  ({TOTAL_SAMPLES//CHUNK_SIZE} chunks × {CHUNK_SIZE:,})\")\nprint(f\"   TARGET_CANCER_RATIO : auto (calculated from data in Cell 6)\")\nprint(f\"   LABEL_SMOOTHING     : {LABEL_SMOOTHING}\")\nprint(f\"   Est. RAM per chunk  : ~{CHUNK_SIZE*IMG_SIZE*IMG_SIZE*3/1e6:.0f} MB (raw)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:17:39.421284Z","iopub.execute_input":"2026-05-10T19:17:39.422052Z","iopub.status.idle":"2026-05-10T19:17:39.428299Z","shell.execute_reply.started":"2026-05-10T19:17:39.422022Z","shell.execute_reply":"2026-05-10T19:17:39.427608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ════════════════════════════════════════════════════════════════\n# CELL 3 — Preprocessing Helper (DCM-aware)\n# ════════════════════════════════════════════════════════════════\nimport pydicom\nclahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n\ndef preprocess_image_cv2(img_path: str, laterality: str) -> np.ndarray:\n    try:\n        dcm = pydicom.dcmread(img_path)\n        img = dcm.pixel_array.astype(np.float32)\n        if hasattr(dcm, 'PhotometricInterpretation') and \\\n           'MONOCHROME1' in str(dcm.PhotometricInterpretation):\n            img = img.max() - img\n        img = ((img - img.min()) / (img.max() - img.min() + 1e-6) * 255).astype(np.uint8)\n    except Exception:\n        return None\n    mask = img > 0\n    rows, cols = mask.any(1), mask.any(0)\n    if not rows.any() or not cols.any():\n        return None\n    img = img[np.ix_(rows, cols)]\n    img = clahe.apply(img)\n    if laterality == 'R':\n        img = cv2.flip(img, 1)\n    img = cv2.cvtColor(img, cv2.COLOR_GRAY2RGB)\n    return img\n\nprint('✅ Preprocessing helper ready! (DCM)')\n","metadata":{"id":"cUusWUlj2MuQ","outputId":"51d3f3e6-5fbd-44fc-f725-3ef794108344","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:21:44.212208Z","iopub.execute_input":"2026-05-10T19:21:44.21339Z","iopub.status.idle":"2026-05-10T19:21:44.220708Z","shell.execute_reply.started":"2026-05-10T19:21:44.213352Z","shell.execute_reply":"2026-05-10T19:21:44.219926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 4 — Dataset Statistics ✅ v12\n# ✅ cache_loader.train_chunks و val_chunks بدل chunks القديمة\n\nprint(f\"Total samples selected : {TOTAL_SAMPLES:,}\")\nprint(f\"Chunk size             : {CHUNK_SIZE:,}\")\nprint(f\"Num chunks             : {cache_loader.num_chunks}\")\nprint(f\"Cancer ratio target    : {CANCER_RATIO:.0%}\")\nprint(f\"Val per chunk          : {N_VAL_PER_CHUNK:,}\")\nprint()\nprint(f\"{'Chunk':<8} {'Train':>8} {'Tr-Cancer':>10} {'Tr-Healthy':>11} {'Tr-Ratio':>9} | {'Val':>6} {'Val-C':>6} {'Val-H':>6} {'Val-Ratio':>9}\")\nprint(\"-\"*80)\nfor i, (tr, vl) in enumerate(zip(cache_loader.train_chunks, cache_loader.val_chunks)):\n    nc_tr = (tr['cancer']==1).sum()\n    nh_tr = (tr['cancer']==0).sum()\n    nc_vl = (vl['cancer']==1).sum()\n    nh_vl = (vl['cancer']==0).sum()\n    print(f\"#{i+1:<7} {len(tr):>8,} {nc_tr:>10,} {nh_tr:>11,} {nc_tr/len(tr):>9.1%} | \"\n          f\"{len(vl):>6,} {nc_vl:>6,} {nh_vl:>6,} {nc_vl/len(vl):>9.1%}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:21:48.914351Z","iopub.execute_input":"2026-05-10T19:21:48.914869Z","iopub.status.idle":"2026-05-10T19:21:48.92663Z","shell.execute_reply.started":"2026-05-10T19:21:48.91484Z","shell.execute_reply":"2026-05-10T19:21:48.925705Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 5 — Class Weight ✅ v13\n# ✅ بناخد cancer_ratio من cache_loader مباشرة (اتحسب تلقائياً في Cell 6)\n\nn_c_per    = cache_loader.n_c_per_chunk\nn_h_per    = cache_loader.n_h_per_chunk\nn_val_c    = cache_loader.n_val_c\nn_val_h    = cache_loader.n_val_h\nn_tr_c     = cache_loader.n_train_c\nn_tr_h     = cache_loader.n_train_h\n\nraw_weight   = n_h_per / max(n_c_per, 1)\nclass_weight = {0: 1.0, 1: float(max(raw_weight, 2.5))}\n\nprint(f\"✅ Class Weight (v13):\")\nprint(f\"   Cancer per chunk  : {n_c_per:,} ({n_c_per/CHUNK_SIZE:.1%}) — from data\")\nprint(f\"   Healthy per chunk : {n_h_per:,} ({n_h_per/CHUNK_SIZE:.1%})\")\nprint(f\"   raw_weight        : {raw_weight:.3f}\")\nprint(f\"   class_weight[1]   : {class_weight[1]:.2f}\")\nprint()\nprint(f\"   Train per chunk   : {n_tr_c:,}c + {n_tr_h:,}h = {n_tr_c+n_tr_h:,}  (cancer ratio: {n_tr_c/(n_tr_c+n_tr_h):.1%})\")\nprint(f\"   Val   per chunk   : {n_val_c:,}c + {n_val_h:,}h = {n_val_c+n_val_h:,}  (cancer ratio: {n_val_c/(n_val_c+n_val_h):.1%})\")\nprint(f\"   Total chunks      : {cache_loader.num_chunks}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:21:54.951414Z","iopub.execute_input":"2026-05-10T19:21:54.952016Z","iopub.status.idle":"2026-05-10T19:21:54.958729Z","shell.execute_reply.started":"2026-05-10T19:21:54.951985Z","shell.execute_reply":"2026-05-10T19:21:54.957764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 6 — RAM Status ✅ v9\nraw_mb    = CHUNK_SIZE * IMG_SIZE * IMG_SIZE * 3 / 1e6\nfloat_mb  = raw_mb * 4\ntotal_est = float_mb + 2000   # model + TF overhead\n\nprint(f\"✅ RAM Estimate per chunk:\")\nprint(f\"   Raw images (uint8)   : ~{raw_mb:.0f} MB\")\nprint(f\"   After TF float32     : ~{float_mb:.0f} MB\")\nprint(f\"   Model + TF overhead  : ~2,000 MB\")\nprint(f\"   Total estimated      : ~{total_est:.0f} MB\")\nprint(f\"   (well within 30 GB limit ✅)\")\nprint()\nprint(f\"✅ Strategy:\")\nprint(f\"   - Load chunk into RAM (uint8)\")\nprint(f\"   - Train → TF converts to float32 on-the-fly per batch\")\nprint(f\"   - After chunk: numpy arrays deleted + tf.clear_session()\")\nprint(f\"   - Next chunk loaded into clean RAM\")\n","metadata":{"id":"CSHMA6sQ2MuR","outputId":"50f7abe5-2945-48b9-9b8c-ad5a1f37b299","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:22:01.154196Z","iopub.execute_input":"2026-05-10T19:22:01.154481Z","iopub.status.idle":"2026-05-10T19:22:01.160165Z","shell.execute_reply.started":"2026-05-10T19:22:01.154458Z","shell.execute_reply":"2026-05-10T19:22:01.159279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 7 — tf.data Pipeline ✅ v12\n# ✅ preprocess_input أول خطوة دايماً\n# ✅ Cancer oversampling 2x — shuffle buffer كافي\n# ✅ augment بعد preprocess (range صح)\nAUTOTUNE = tf.data.AUTOTUNE\n\ndef preprocess_tf(img, lbl):\n    img = tf.cast(img, tf.float32)\n    img = tf.keras.applications.efficientnet.preprocess_input(img)\n    return img, tf.cast(lbl, tf.float32)\n\ndef augment(img, lbl):\n    # Geometric\n    img = tf.image.random_flip_left_right(img)\n    img = tf.image.random_flip_up_down(img)\n    k   = tf.random.uniform([], 0, 4, dtype=tf.int32)\n    img = tf.image.rot90(img, k=k)\n    # Intensity (range بعد preprocess_input هو [-1, 1])\n    img = tf.image.random_brightness(img, 0.15)\n    img = tf.image.random_contrast(img, 0.80, 1.20)\n    img = tf.clip_by_value(img, -1.0, 1.0)\n    # Random crop + resize\n    crop_frac = tf.random.uniform([], 0.85, 1.0)\n    crop_h    = tf.cast(crop_frac * IMG_SIZE, tf.int32)\n    crop_w    = tf.cast(crop_frac * IMG_SIZE, tf.int32)\n    img       = tf.image.random_crop(img, [crop_h, crop_w, 3])\n    img       = tf.image.resize(img, [IMG_SIZE, IMG_SIZE])\n    # Gaussian noise\n    noise = tf.random.normal(tf.shape(img), 0.0, 0.012)\n    img   = tf.clip_by_value(img + noise, -1.0, 1.0)\n    return img, lbl\n\n\ndef make_dataset_from_cache(imgs_np, labels_np, training=False):\n    \"\"\"\n    ✅ v12: Cancer oversampling 2x داخل الـ pipeline\n    ✅ shuffle buffer = حجم الداتا الكامل بعد الـ oversampling\n    \"\"\"\n    if training:\n        cancer_mask = labels_np == 1\n        n_cancer    = int(cancer_mask.sum())\n        n_healthy   = int((~cancer_mask).sum())\n\n        if n_cancer > 0 and n_healthy > 0:\n            ds_cancer  = tf.data.Dataset.from_tensor_slices(\n                (imgs_np[cancer_mask], labels_np[cancer_mask]))\n            ds_healthy = tf.data.Dataset.from_tensor_slices(\n                (imgs_np[~cancer_mask], labels_np[~cancer_mask]))\n            ds = ds_cancer.concatenate(ds_cancer).concatenate(ds_healthy)\n            shuffle_buf = n_cancer * 2 + n_healthy   # كامل — مش مقطوع\n        else:\n            ds = tf.data.Dataset.from_tensor_slices((imgs_np, labels_np))\n            shuffle_buf = len(imgs_np)\n\n        ds = ds.shuffle(shuffle_buf, reshuffle_each_iteration=True)\n    else:\n        ds = tf.data.Dataset.from_tensor_slices((imgs_np, labels_np))\n\n    ds = ds.map(preprocess_tf, num_parallel_calls=AUTOTUNE)\n    if training:\n        ds = ds.map(augment, num_parallel_calls=AUTOTUNE)\n    ds = ds.batch(BATCH).prefetch(AUTOTUNE)\n    return ds\n\nprint(\"✅ tf.data pipeline v12 ready!\")\nprint(\"   ✅ Cancer 2x oversampling — full shuffle buffer\")\nprint(\"   ✅ preprocess_input → augment — order correct\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:22:11.757167Z","iopub.execute_input":"2026-05-10T19:22:11.757685Z","iopub.status.idle":"2026-05-10T19:22:11.768823Z","shell.execute_reply.started":"2026-05-10T19:22:11.757656Z","shell.execute_reply":"2026-05-10T19:22:11.767986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 8 — Sample Batch Preview ✅ v6\n# مش هنعرض batch دلوقتي — الـ images بتتحمل في الـ Incremental Loop\nprint(\"✅ Batch preview will be available after first chunk loads in Cell 26\")","metadata":{"id":"bTBpyCHw2MuR","outputId":"24a60093-4c4d-49be-ad99-85e030832c01","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:22:24.67212Z","iopub.execute_input":"2026-05-10T19:22:24.672708Z","iopub.status.idle":"2026-05-10T19:22:24.676664Z","shell.execute_reply.started":"2026-05-10T19:22:24.672678Z","shell.execute_reply":"2026-05-10T19:22:24.67593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ════════════════════════════════════════════════════════════════\n# CELL 9 — Focal Loss  ✅ v8\n# gamma=3.0  → يركز أكتر على الـ hard cancer cases\n# alpha=0.90 → penalty أعلى لو فاتته cancer (miss)\n# ════════════════════════════════════════════════════════════════\nclass FocalLoss(keras.losses.Loss):\n    def __init__(self, gamma=3.0, alpha=0.90, smoothing=0.05, **kwargs):\n        super().__init__(**kwargs)\n        self.gamma     = gamma\n        self.alpha     = alpha\n        self.smoothing = smoothing\n\n    def call(self, y_true, y_pred):\n        y_pred   = tf.reshape(tf.cast(y_pred,  tf.float32), [-1])\n        y_true   = tf.reshape(tf.cast(y_true,  tf.float32), [-1])\n        y_true_s = y_true * (1.0 - self.smoothing) + 0.5 * self.smoothing\n        prob     = tf.sigmoid(y_pred)\n        bce      = tf.nn.sigmoid_cross_entropy_with_logits(labels=y_true_s, logits=y_pred)\n        p_t      = y_true * prob + (1.0 - y_true) * (1.0 - prob)\n        alpha_t  = y_true * self.alpha + (1.0 - y_true) * (1.0 - self.alpha)\n        loss     = alpha_t * tf.pow(1.0 - p_t, self.gamma) * bce\n        return tf.reduce_mean(loss)\n\n    def get_config(self):\n        cfg = super().get_config()\n        cfg.update({\"gamma\": self.gamma, \"alpha\": self.alpha, \"smoothing\": self.smoothing})\n        return cfg\n\nprint(\"✅ FocalLoss v8 — gamma=3.0, alpha=0.90 (optimized for AUC)\")\n","metadata":{"id":"9F0KnnA32MuS","outputId":"ea09275f-19dc-4755-d43d-f99495761371","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:22:29.780641Z","iopub.execute_input":"2026-05-10T19:22:29.781187Z","iopub.status.idle":"2026-05-10T19:22:29.789064Z","shell.execute_reply.started":"2026-05-10T19:22:29.781154Z","shell.execute_reply":"2026-05-10T19:22:29.787965Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ════════════════════════════════════════════════════════════════\n# CELL 10 — EfficientNetB4 + GeM + CBAM Attention  ✅ v5\n# ════════════════════════════════════════════════════════════════\nclass GeM(layers.Layer):\n    def __init__(self, p=3.0, eps=1e-6, **kwargs):\n        super().__init__(**kwargs)\n        self.p   = tf.Variable(p, trainable=True, dtype=tf.float32, name='gem_p')\n        self.eps = eps\n\n    def call(self, x):\n        x = tf.cast(x, tf.float32)\n        x = tf.clip_by_value(x, self.eps, tf.reduce_max(x))\n        x = tf.pow(x, self.p)\n        x = tf.reduce_mean(x, axis=[1, 2])\n        x = tf.pow(x, 1.0 / self.p)\n        return x\n\n    def get_config(self):\n        cfg = super().get_config()\n        cfg.update({'p': float(self.p.numpy()), 'eps': self.eps})\n        return cfg\n\n\ndef build_model(img_size=IMG_SIZE, dropout_rate=0.40):\n    base = tf.keras.applications.EfficientNetB4(\n        include_top=False,\n        weights=\"imagenet\",\n        input_shape=(img_size, img_size, 3)\n    )\n    base.trainable = False\n\n    inputs = keras.Input(shape=(img_size, img_size, 3))\n    x      = base(inputs, training=False)\n\n    # ✅ GeM Pooling\n    x = GeM(p=3.0, name='gem_pool')(x)\n\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(dropout_rate)(x)\n    x = layers.Dense(512, kernel_regularizer=keras.regularizers.l2(1e-4))(x)\n    x = layers.Activation('swish')(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(dropout_rate * 0.5)(x)\n    x = layers.Dense(256, kernel_regularizer=keras.regularizers.l2(1e-4))(x)\n    x = layers.Activation('swish')(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(dropout_rate * 0.25)(x)\n\n    outputs = layers.Dense(1, dtype='float32')(x)\n    model = keras.Model(inputs, outputs, name=\"BreastCancer_B4_GeM_v5\")\n    return model, base\n\n\nmodel, base_model = build_model()\nmodel.summary()\nprint(f\"\\n✅ EfficientNetB4 + GeM v5 built!\")\nprint(f\"   Total params    : {model.count_params():,}\")\n","metadata":{"id":"h8GjDxah2MuS","outputId":"a6873e18-8c0f-468d-8031-b92a1af43798","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:22:34.89714Z","iopub.execute_input":"2026-05-10T19:22:34.897693Z","iopub.status.idle":"2026-05-10T19:22:42.732384Z","shell.execute_reply.started":"2026-05-10T19:22:34.897664Z","shell.execute_reply":"2026-05-10T19:22:42.731528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport keras\n\n# ══════════════════════════════════════════════\n# WarmupCosineDecay\n# ══════════════════════════════════════════════\nclass WarmupCosineDecay(keras.optimizers.schedules.LearningRateSchedule):\n    def __init__(self, peak_lr, warmup_steps, total_steps, min_lr=1e-7):\n        super().__init__()\n        self.peak_lr      = float(peak_lr)\n        self.warmup_steps = float(warmup_steps)\n        self.total_steps  = float(total_steps)\n        self.min_lr       = float(min_lr)\n\n    def __call__(self, step):\n        step      = tf.cast(step, tf.float32)\n        warmup_lr = self.peak_lr * (step / tf.maximum(self.warmup_steps, 1.0))\n        progress  = (step - self.warmup_steps) / tf.maximum(\n                        self.total_steps - self.warmup_steps, 1.0)\n        cosine_lr = self.min_lr + 0.5 * (self.peak_lr - self.min_lr) * (\n                        1.0 + tf.cos(3.14159265 * progress))\n        return tf.where(step < self.warmup_steps, warmup_lr, cosine_lr)\n\n    def get_config(self):\n        return dict(peak_lr=self.peak_lr, warmup_steps=self.warmup_steps,\n                    total_steps=self.total_steps, min_lr=self.min_lr)\n\nprint(\"✅ WarmupCosineDecay defined\")\n\n# ══════════════════════════════════════════════\n# CELL 11 — Phase 1: Head Training  ✅ v13\n# ══════════════════════════════════════════════\nimport gc\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\n\n# ✅ FIX: من cache_loader مباشرة\n_cancer_ratio = cache_loader.cancer_ratio\n_n_tmp_c = round(2000 * _cancer_ratio)\n_n_tmp_h = 2000 - _n_tmp_c\n_avail_c = len(df_train[df_train[\"cancer\"] == 1])\n_n_tmp_c = min(_n_tmp_c, _avail_c)\n\n_df_tmp = pd.concat([\n    df_train[df_train[\"cancer\"] == 1].sample(_n_tmp_c, random_state=999),\n    df_train[df_train[\"cancer\"] == 0].sample(_n_tmp_h, random_state=999),\n]).sample(frac=1, random_state=999).reset_index(drop=True)\n\nprint(f\"Loading temp sample for Phase 1: {len(_df_tmp):,} images \"\n      f\"(cancer={_n_tmp_c:,} | healthy={_n_tmp_h:,}) ...\")\n\n# ✅ FIX: _load_df v13 بترجع 2 values بس\n_tmp_imgs, _tmp_lbls = cache_loader._load_df(_df_tmp)\n\n_tr_idx, _vl_idx = train_test_split(\n    np.arange(len(_tmp_imgs)), test_size=0.2,\n    random_state=42, stratify=_tmp_lbls\n)\ntrain_ds = make_dataset_from_cache(_tmp_imgs[_tr_idx], _tmp_lbls[_tr_idx], training=True)\nval_ds   = make_dataset_from_cache(_tmp_imgs[_vl_idx], _tmp_lbls[_vl_idx], training=False)\nprint(f\"   train={len(_tr_idx):,} | val={len(_vl_idx):,}\")\nprint(\"Phase 1: Training head only — base frozen\\n\")\n\nP1_EPOCHS      = 15\np1_total_steps = len(train_ds) * P1_EPOCHS\np1_warmup      = len(train_ds) * WARMUP_EPOCHS\n\np1_lr = WarmupCosineDecay(\n    peak_lr=5e-4, warmup_steps=p1_warmup,\n    total_steps=p1_total_steps, min_lr=1e-5\n)\n\nmodel.compile(\n    optimizer=keras.optimizers.Adam(p1_lr),\n    loss=FocalLoss(gamma=3.0, alpha=0.90, smoothing=LABEL_SMOOTHING),\n    metrics=[\n        keras.metrics.BinaryAccuracy(name=\"accuracy\", threshold=0.0),\n        keras.metrics.AUC(name=\"auc\",                 from_logits=True),\n        keras.metrics.Precision(name=\"precision\",     thresholds=0.0),\n        keras.metrics.Recall(name=\"recall\",           thresholds=0.0),\n    ]\n)\n\nphase1_cb = [\n    callbacks.EarlyStopping(\n        monitor=\"val_auc\", patience=6,\n        restore_best_weights=True, mode=\"max\", verbose=1),\n    callbacks.ModelCheckpoint(\n        filepath=os.path.join(SAVE_DIR, \"best_phase1.keras\"),\n        monitor=\"val_auc\", save_best_only=True, mode=\"max\", verbose=1),\n]\n\nh1 = model.fit(\n    train_ds, validation_data=val_ds,\n    epochs=P1_EPOCHS, class_weight=class_weight, callbacks=phase1_cb\n)\n\ndel _tmp_imgs, _tmp_lbls, _df_tmp, train_ds, val_ds\ngc.collect()\nprint(f\"\\n✅ Phase 1 done! Best val_auc : {max(h1.history['val_auc']):.4f}\")\nprint(f\"   Best val_acc              : {max(h1.history['val_accuracy']):.4f}\")\nprint(\"✅ Temp images cleared — cache_loader untouched\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:22:50.273451Z","iopub.execute_input":"2026-05-10T19:22:50.274264Z","iopub.status.idle":"2026-05-10T19:54:36.36192Z","shell.execute_reply.started":"2026-05-10T19:22:50.27423Z","shell.execute_reply":"2026-05-10T19:54:36.361319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ════════════════════════════════════════════════════════════════\n# CELL 12 — Phase 2: Fine-tuning  ✅ v13\n# ════════════════════════════════════════════════════════════════\nimport gc\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\n\nprint(\"Phase 2: Fine-tuning top 80 layers\")\n\n# ✅ FIX: من cache_loader مباشرة\n_cancer_ratio2 = cache_loader.cancer_ratio\n_n_tmp_c2 = round(2000 * _cancer_ratio2)\n_n_tmp_h2 = 2000 - _n_tmp_c2\n_avail_c2 = len(df_train[df_train[\"cancer\"] == 1])\n_n_tmp_c2 = min(_n_tmp_c2, _avail_c2)\n\n_df_tmp2 = pd.concat([\n    df_train[df_train[\"cancer\"] == 1].sample(_n_tmp_c2, random_state=777),\n    df_train[df_train[\"cancer\"] == 0].sample(_n_tmp_h2, random_state=777),\n]).sample(frac=1, random_state=777).reset_index(drop=True)\n\nprint(f\"Loading temp sample for Phase 2: {len(_df_tmp2):,} images \"\n      f\"(cancer={_n_tmp_c2:,} | healthy={_n_tmp_h2:,}) ...\")\n\n# ✅ FIX: _load_df v13 بترجع 2 values بس\n_tmp_imgs2, _tmp_lbls2 = cache_loader._load_df(_df_tmp2)\n\n_tr_idx2, _vl_idx2 = train_test_split(\n    np.arange(len(_tmp_imgs2)), test_size=0.2,\n    random_state=42, stratify=_tmp_lbls2\n)\ntrain_ds = make_dataset_from_cache(\n    _tmp_imgs2[_tr_idx2], _tmp_lbls2[_tr_idx2], training=True)\nval_ds   = make_dataset_from_cache(\n    _tmp_imgs2[_vl_idx2], _tmp_lbls2[_vl_idx2], training=False)\nprint(f\"   train={len(_tr_idx2):,} | val={len(_vl_idx2):,}\")\n\n# ── Unfreeze top 80 layers ────────────────────────────────────\nbase_model.trainable = True\nfor layer in base_model.layers[:-80]:\n    layer.trainable = False\nfor layer in base_model.layers:\n    if isinstance(layer, layers.BatchNormalization):\n        layer.trainable = False\n\ntrainable_count = sum(np.prod(v.shape) for v in model.trainable_variables)\nprint(f\"   Trainable params : {trainable_count:,}\")\n\np2_total_steps = len(train_ds) * EPOCHS\np2_warmup      = len(train_ds) * 1\n\np2_lr = WarmupCosineDecay(\n    peak_lr=2e-5, warmup_steps=p2_warmup,\n    total_steps=p2_total_steps, min_lr=1e-7\n)\n\nmodel.compile(\n    optimizer=keras.optimizers.Adam(p2_lr),\n    loss=FocalLoss(gamma=2.5, alpha=0.85, smoothing=LABEL_SMOOTHING),\n    metrics=[\n        keras.metrics.BinaryAccuracy(name=\"accuracy\", threshold=0.0),\n        keras.metrics.AUC(name=\"auc\",                 from_logits=True),\n        keras.metrics.Precision(name=\"precision\",     thresholds=0.0),\n        keras.metrics.Recall(name=\"recall\",           thresholds=0.0),\n    ]\n)\n\nphase2_cb = [\n    callbacks.EarlyStopping(\n        monitor=\"val_auc\", patience=8,\n        restore_best_weights=True, mode=\"max\", verbose=1),\n    callbacks.ModelCheckpoint(\n        filepath=os.path.join(SAVE_DIR, \"best_model.keras\"),\n        monitor=\"val_auc\", save_best_only=True, mode=\"max\", verbose=1),\n    callbacks.CSVLogger(os.path.join(SAVE_DIR, \"training_log.csv\")),\n]\n\nh2 = model.fit(\n    train_ds, validation_data=val_ds,\n    epochs=EPOCHS, class_weight=class_weight, callbacks=phase2_cb\n)\n\ndel _tmp_imgs2, _tmp_lbls2, _df_tmp2, train_ds, val_ds\ngc.collect()\ntry:\n    tf.keras.backend.clear_session()\n    gc.collect()\nexcept Exception:\n    pass\n\nprint(f\"\\n✅ Phase 2 done! Best val_auc : {max(h2.history['val_auc']):.4f}\")\nprint(f\"   Best val_acc              : {max(h2.history['val_accuracy']):.4f}\")\nprint(\"✅ RAM fully cleared — cache_loader ready for incremental loop\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T20:27:33.401758Z","iopub.execute_input":"2026-05-10T20:27:33.402611Z","execution_failed":"2026-05-10T22:21:39.118Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n\n# **Incremental Training —40000  صورة / 6 Chunks × 5,000**\n### الـ model بيتدرب على كل chunk ويمسحه قبل ما يحمل الجديد\n","metadata":{}},{"cell_type":"code","source":"# ════════════════════════════════════════════════════════════════\n# CELL — Incremental Training Loop  ✅ v14\n#\n# ✅ FIX: recompile بـ float LR قبل الـ loop عشان ReduceLROnPlateau يشتغل\n# ✅ FIX: clear_current_chunk() بيمسح الـ arrays صراحةً من RAM\n# ✅ FIX: next_chunk() بيلود الجديد بعد المسح مباشرة\n# ✅ val_imgs_all بيتجمع للـ TTA بعد الـ loop\n# ✅ ModelCheckpoint يحفظ الأحسن AUC\n# ════════════════════════════════════════════════════════════════\nimport gc\nimport os\nimport numpy as np\nimport tensorflow as tf\nimport keras\n\nEPOCHS_PER_CHUNK  = 7\nAUTOTUNE          = tf.data.AUTOTUNE\nINC_LR_INITIAL    = 2e-5   # LR ابتدائية للـ incremental — أقل من phase 2\n\nval_imgs_all   = []\nval_labels_all = []\nbest_inc_auc   = 0.0\n\n\n# ══════════════════════════════════════════════════════════════\n#  ✅ FIX: recompile بـ float LR عشان ReduceLROnPlateau يشتغل\n#  الـ WarmupCosineDecay اللي اتبنى في Phase 2 مش settable\n#  لازم نعمل compile جديد بـ float عشان الـ callback يقدر يعدل\n# ══════════════════════════════════════════════════════════════\nprint(\"🔧 Recompiling model with float LR for incremental loop...\")\nmodel.compile(\n    optimizer=keras.optimizers.Adam(learning_rate=INC_LR_INITIAL),\n    loss=FocalLoss(gamma=2.5, alpha=0.85, smoothing=LABEL_SMOOTHING),\n    metrics=[\n        keras.metrics.BinaryAccuracy(name=\"accuracy\", threshold=0.0),\n        keras.metrics.AUC(name=\"auc\",                 from_logits=True),\n        keras.metrics.Precision(name=\"precision\",     thresholds=0.0),\n        keras.metrics.Recall(name=\"recall\",           thresholds=0.0),\n    ]\n)\nprint(f\"✅ Recompiled — Adam(lr={INC_LR_INITIAL:.0e}) — ReduceLROnPlateau will work now\\n\")\n\n\n# ══════════════════════════════════════════════════════════════\n#  tf.data pipeline\n# ══════════════════════════════════════════════════════════════\ndef make_inc_dataset(images, labels, training=False):\n    \"\"\"tf.data pipeline ✅ v14\"\"\"\n    def preprocess(img, lbl):\n        img = tf.cast(img, tf.float32)\n        img = tf.keras.applications.efficientnet.preprocess_input(img)\n        return img, tf.cast(lbl, tf.float32)\n\n    def augment(img, lbl):\n        img = tf.image.random_flip_left_right(img)\n        img = tf.image.random_flip_up_down(img)\n        k   = tf.random.uniform([], 0, 4, dtype=tf.int32)\n        img = tf.image.rot90(img, k=k)\n        img = tf.image.random_brightness(img, 0.15)\n        img = tf.image.random_contrast(img, 0.80, 1.20)\n        img = tf.clip_by_value(img, -1.0, 1.0)\n        crop_frac = tf.random.uniform([], 0.85, 1.0)\n        crop_h    = tf.cast(crop_frac * IMG_SIZE, tf.int32)\n        crop_w    = tf.cast(crop_frac * IMG_SIZE, tf.int32)\n        img       = tf.image.random_crop(img, [crop_h, crop_w, 3])\n        img       = tf.image.resize(img, [IMG_SIZE, IMG_SIZE])\n        noise     = tf.random.normal(tf.shape(img), 0.0, 0.012)\n        img       = tf.clip_by_value(img + noise, -1.0, 1.0)\n        return img, lbl\n\n    ds = tf.data.Dataset.from_tensor_slices((images, labels))\n    if training:\n        ds = ds.shuffle(len(images), reshuffle_each_iteration=True)\n    ds = ds.map(preprocess, num_parallel_calls=AUTOTUNE)\n    if training:\n        ds = ds.map(augment, num_parallel_calls=AUTOTUNE)\n    ds = ds.batch(BATCH).prefetch(AUTOTUNE)\n    return ds\n\n\n# ══════════════════════════════════════════════════════════════\n#  Cumulative Validation\n# ══════════════════════════════════════════════════════════════\ndef run_cumulative_val(model, loader, tracker, chunk_id):\n    \"\"\"Inference على val set + تحديث cumulative tracker\"\"\"\n    val_ds = make_inc_dataset(loader.val_images, loader.val_labels, training=False)\n    probs_l, labels_l = [], []\n    for imgs_b, lbls_b in val_ds:\n        p = tf.sigmoid(model(imgs_b, training=False)).numpy().flatten()\n        probs_l.extend(p.tolist())\n        labels_l.extend(lbls_b.numpy().tolist())\n    del val_ds\n    tracker.add_chunk_results(\n        np.array(labels_l, dtype=np.int32),\n        np.array(probs_l,  dtype=np.float32),\n        chunk_id\n    )\n    tracker.print_summary(chunk_id)\n    return tracker.cumulative_auc()\n\n\n# ══════════════════════════════════════════════════════════════\n#  ✅ FIX: مسح الـ chunk من RAM صراحةً\n# ══════════════════════════════════════════════════════════════\ndef clear_chunk_from_ram(loader):\n    \"\"\"بيمسح train/val arrays من loader صراحةً عشان Python يحرر الـ RAM\"\"\"\n    for attr in (\"train_images\", \"train_labels\", \"val_images\", \"val_labels\"):\n        if getattr(loader, attr, None) is not None:\n            del loader.__dict__[attr]\n            setattr(loader, attr, None)\n    gc.collect()\n    print(\"   🧹 Chunk cleared from RAM\")\n\n\n# ══════════════════════════════════════════════════════════════\n#  Main Incremental Loop\n# ══════════════════════════════════════════════════════════════\nprint(\"🚀 Incremental Training  ✅ v14\")\nprint(f\"   Total chunks     : {cache_loader.num_chunks}\")\nprint(f\"   Images per chunk : {CHUNK_SIZE:,}  \"\n      f\"(cancer={cache_loader.n_c_per_chunk:,} | ratio={cache_loader.cancer_ratio:.1%})\")\nprint(f\"   Epochs per chunk : {EPOCHS_PER_CHUNK}\")\nprint(f\"   Initial LR       : {INC_LR_INITIAL:.0e}  (ReduceLROnPlateau enabled)\")\nprint(f\"   Continues from Phase 2 weights\\n\")\n\ninc_histories = []\n\n# ── اللود الأول ─────────────────────────────────────────────\nchunk_id = cache_loader.next_chunk()\n\nwhile chunk_id is not None:\n    print(f\"\\n{'#'*65}\")\n    print(f\"  🔥 Chunk #{chunk_id}/{cache_loader.num_chunks}\")\n\n    # ── Guard: skip إذا فاضي ────────────────────────────────\n    if cache_loader.train_images is None or len(cache_loader.train_images) == 0:\n        print(f\"  ⚠️ Chunk #{chunk_id} empty — skipping!\")\n        clear_chunk_from_ram(cache_loader)\n        chunk_id = cache_loader.next_chunk()\n        continue\n\n    n_c_tr   = (cache_loader.train_labels == 1).sum()\n    n_h_tr   = (cache_loader.train_labels == 0).sum()\n    n_c_vl   = (cache_loader.val_labels   == 1).sum()\n    n_h_vl   = (cache_loader.val_labels   == 0).sum()\n    tr_ratio = n_c_tr / max(len(cache_loader.train_labels), 1)\n    vl_ratio = n_c_vl / max(len(cache_loader.val_labels),   1)\n\n    print(f\"  Train: {len(cache_loader.train_images):,}  \"\n          f\"cancer={n_c_tr:,} ({tr_ratio:.1%}) | healthy={n_h_tr:,}\")\n    print(f\"  Val  : {len(cache_loader.val_images):,}  \"\n          f\"cancer={n_c_vl:,} ({vl_ratio:.1%}) | healthy={n_h_vl:,}\")\n\n    if n_c_tr == 0:\n        print(f\"  ⚠️ Chunk #{chunk_id} has no cancer samples — skipping!\")\n        clear_chunk_from_ram(cache_loader)\n        chunk_id = cache_loader.next_chunk()\n        continue\n\n    print(f\"{'#'*65}\")\n\n    # ── Build tf.data ────────────────────────────────────────\n    train_ds_inc = make_inc_dataset(\n        cache_loader.train_images,\n        cache_loader.train_labels,\n        training=True\n    )\n\n    # ── Training ─────────────────────────────────────────────\n    history = model.fit(\n        train_ds_inc,\n        epochs  = EPOCHS_PER_CHUNK,\n        verbose = 1,\n        callbacks=[\n            # ✅ يشتغل دلوقتي لأن الـ optimizer اتعمل بـ float LR\n            tf.keras.callbacks.ReduceLROnPlateau(\n                monitor=\"loss\",\n                factor=0.5,\n                patience=2,\n                min_lr=1e-7,\n                verbose=1\n            ),\n        ]\n    )\n    inc_histories.append(history.history)\n\n    # ── مسح train dataset فورًا ──────────────────────────────\n    del train_ds_inc\n    gc.collect()\n\n    # ── Validation (قبل مسح الـ chunk) ──────────────────────\n    chunk_auc = run_cumulative_val(model, cache_loader, val_tracker, chunk_id)\n\n    # ── حفظ val للـ TTA (copy قبل المسح) ────────────────────\n    val_imgs_all.append(cache_loader.val_images.copy())\n    val_labels_all.append(cache_loader.val_labels.copy())\n\n    # ── Checkpointing ────────────────────────────────────────\n    if chunk_auc > best_inc_auc:\n        best_inc_auc = chunk_auc\n        model.save(os.path.join(SAVE_DIR, \"best_model.keras\"))\n        print(f\"   💾 New best AUC={best_inc_auc:.4f} — saved best_model.keras\")\n\n    model.save(os.path.join(SAVE_DIR, f\"model_chunk{chunk_id:02d}.keras\"))\n    print(f\"   💾 Chunk checkpoint: model_chunk{chunk_id:02d}.keras\")\n\n    # ══════════════════════════════════════════════════════════\n    #  ✅ FIX: مسح الـ chunk الحالي من RAM قبل لود الجديد\n    #  الترتيب مهم: امسح الأول → بعدين لود التاني\n    # ══════════════════════════════════════════════════════════\n    clear_chunk_from_ram(cache_loader)   # ← مسح arrays القديمة\n    chunk_id = cache_loader.next_chunk() # ← لود الجديد في RAM نظيفة\n\n\n# ── Build val arrays للـ TTA ────────────────────────────────\nval_imgs_arr = np.concatenate(val_imgs_all,  axis=0)\nval_labels   = np.concatenate(val_labels_all, axis=0).astype(int)\n\ndf_val = pd.DataFrame({\n    'cancer'    : val_labels,\n    'patient_id': ['unk'] * len(val_labels),\n    'image_id'  : ['unk'] * len(val_labels),\n    'laterality': ['L']   * len(val_labels),\n})\n\nprint(\"\\n\" + \"=\"*65)\nprint(\"  ✅ Incremental Training Complete!\")\nprint(f\"  Chunks trained       : {cache_loader.num_chunks}\")\nprint(f\"  Total images trained : {cache_loader.num_chunks * CHUNK_SIZE:,}\")\nprint(f\"  Val images collected : {len(val_imgs_arr):,} (for TTA)\")\nprint(f\"  Cancer in val        : {(val_labels==1).sum():,} ({(val_labels==1).mean():.1%})\")\nprint(f\"  Healthy in val       : {(val_labels==0).sum():,} ({(val_labels==0).mean():.1%})\")\nprint(f\"  Final Cumul. Accuracy: {val_tracker.cumulative_accuracy():.4f} \"\n      f\"({val_tracker.cumulative_accuracy()*100:.2f}%)\")\ntry:\n    print(f\"  Final Cumul. AUC     : {val_tracker.cumulative_auc():.4f}\")\nexcept Exception:\n    pass\nprint(f\"  Best Inc. AUC        : {best_inc_auc:.4f}\")\nprint(\"=\"*65)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:16:46.530128Z","iopub.status.idle":"2026-05-10T19:16:46.530476Z","shell.execute_reply.started":"2026-05-10T19:16:46.53031Z","shell.execute_reply":"2026-05-10T19:16:46.530332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ════════════════════════════════════════════════════════════════\n# CELL — Final Cumulative Report + Plot  ✅ v10\n# ════════════════════════════════════════════════════════════════\nimport matplotlib.pyplot as plt\nimport numpy as np\n\nval_tracker.print_summary(cache_loader.num_chunks)\n\nchunk_ids  = [m['chunk_id']    for m in val_tracker.chunk_metrics]\naccs       = [m['acc']         for m in val_tracker.chunk_metrics]\nsens       = [m['sensitivity'] for m in val_tracker.chunk_metrics]\nspec       = [m['specificity'] for m in val_tracker.chunk_metrics]\n\n# حساب الـ cumulative accuracy في كل خطوة\ncumul_accs = []\nfor i in range(len(val_tracker.all_labels)):\n    all_l = np.concatenate(val_tracker.all_labels[:i+1])\n    all_p = np.concatenate(val_tracker.all_probs[:i+1])\n    preds = (all_p >= 0.5).astype(int)\n    cumul_accs.append((preds == all_l).mean())\n\nfig, axes = plt.subplots(1, 2, figsize=(14, 5))\n\n# ── Per-chunk metrics ────────────────────────────────────────\naxes[0].plot(chunk_ids, accs,  'o-', color='#3B8BD4', lw=2, ms=8, label='Per-chunk Acc')\naxes[0].plot(chunk_ids, sens,  's-', color='#E8593C', lw=2, ms=8, label='Sensitivity')\naxes[0].plot(chunk_ids, spec,  '^-', color='#27ae60', lw=2, ms=8, label='Specificity')\naxes[0].set_xlabel('Chunk #', fontsize=12)\naxes[0].set_ylabel('Metric', fontsize=12)\naxes[0].set_title('Per-Chunk Val Metrics', fontsize=13)\naxes[0].legend(fontsize=10)\naxes[0].set_ylim(0, 1)\naxes[0].set_xticks(chunk_ids)\naxes[0].grid(alpha=0.3)\n\n# ── Cumulative accuracy ──────────────────────────────────────\naxes[1].plot(chunk_ids, cumul_accs, 'D-', color='#8e44ad', lw=2.5, ms=10)\naxes[1].fill_between(chunk_ids, cumul_accs, alpha=0.15, color='#8e44ad')\nfor x, y in zip(chunk_ids, cumul_accs):\n    axes[1].annotate(f'{y:.3f}', (x, y),\n                     textcoords=\"offset points\", xytext=(0, 10),\n                     ha='center', fontsize=11, fontweight='bold', color='#8e44ad')\naxes[1].set_xlabel('Chunk #', fontsize=12)\naxes[1].set_ylabel('Cumulative Val Accuracy', fontsize=12)\naxes[1].set_title('📈 Cumulative Val Accuracy (تراكمي)', fontsize=13)\naxes[1].set_ylim(0, 1)\naxes[1].set_xticks(chunk_ids)\naxes[1].grid(alpha=0.3)\n\nplt.suptitle(\n    f'Incremental Learning — {TOTAL_SAMPLES:,} صورة / {cache_loader.num_chunks} Chunks\\n'\n    f'Final Cumulative Accuracy = {val_tracker.cumulative_accuracy():.4f}',\n    fontsize=13\n)\nplt.tight_layout()\nplt.savefig(os.path.join(SAVE_DIR, \"incremental_cumulative_val.png\"), dpi=150)\nplt.show()\n\nprint(f\"\\n🎯 Final Cumulative Accuracy : {val_tracker.cumulative_accuracy():.4f}  \"\n      f\"({val_tracker.cumulative_accuracy()*100:.2f}%)\")\ntry:\n    print(f\"🎯 Final Cumulative AUC     : {val_tracker.cumulative_auc():.4f}\")\nexcept Exception:\n    pass\nprint(f\"✅ Saved: incremental_cumulative_val.png\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:16:46.532262Z","iopub.status.idle":"2026-05-10T19:16:46.532808Z","shell.execute_reply.started":"2026-05-10T19:16:46.532615Z","shell.execute_reply":"2026-05-10T19:16:46.532642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ════════════════════════════════════════════════════════════════\n# CELL 13 — Training History  ✅ v12\n# ════════════════════════════════════════════════════════════════\nacc   = h1.history[\"accuracy\"]      + h2.history[\"accuracy\"]\nvacc  = h1.history[\"val_accuracy\"]  + h2.history[\"val_accuracy\"]\nauc   = h1.history[\"auc\"]           + h2.history[\"auc\"]\nvauc  = h1.history[\"val_auc\"]       + h2.history[\"val_auc\"]\nprec  = h1.history[\"precision\"]     + h2.history[\"precision\"]\nvprec = h1.history[\"val_precision\"] + h2.history[\"val_precision\"]\nrec   = h1.history[\"recall\"]        + h2.history[\"recall\"]\nvrec  = h1.history[\"val_recall\"]    + h2.history[\"val_recall\"]\np1    = len(h1.history[\"accuracy\"])\n\nfig, axes = plt.subplots(2, 2, figsize=(16, 10))\nfor ax, (tr, vl, title) in zip(axes.flat, [\n    (acc,  vacc,  \"Accuracy\"),\n    (auc,  vauc,  \"AUC\"),\n    (prec, vprec, \"Precision\"),\n    (rec,  vrec,  \"Recall\")\n]):\n    ax.plot(tr, label=\"Train\", color=\"#3B8BD4\", lw=2)\n    ax.plot(vl, label=\"Val\",   color=\"#E8593C\", lw=2)\n    ax.axvline(p1, color=\"gray\", ls=\"--\", alpha=0.6, label=\"Fine-tune start\")\n    ax.set_title(title, fontsize=12)\n    ax.legend()\n    ax.set_xlabel(\"Epoch\")\n    ax.grid(alpha=0.3)\n\n# ✅ Title صح — EfficientNetB4 مش B2\nplt.suptitle(\"Training History — EfficientNetB4 + GeM (Phase 1 + Phase 2)\", fontsize=14)\nplt.tight_layout()\nplt.savefig(os.path.join(SAVE_DIR, \"training_history.png\"), dpi=150)\nplt.show()\nprint(\"✅ Plot saved!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:16:46.533867Z","iopub.status.idle":"2026-05-10T19:16:46.534252Z","shell.execute_reply.started":"2026-05-10T19:16:46.534039Z","shell.execute_reply":"2026-05-10T19:16:46.534063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ════════════════════════════════════════════════════════════════\n# CELL 14 — Threshold Tuning + TTA  ✅ v10 FIXED\n#\n# ✅ val_imgs_arr و val_labels تتبنى من الـ Incremental Loop\n# ✅ preprocess_input مش بيتعمل مرتين\n# ✅ TTA augmentations على الـ raw uint8 قبل preprocess\n# ════════════════════════════════════════════════════════════════\nfrom sklearn.metrics import roc_auc_score, roc_curve, precision_recall_curve\nimport numpy as np\n\nmodel = keras.models.load_model(\n    os.path.join(SAVE_DIR, \"best_model.keras\"),\n    custom_objects={'FocalLoss': FocalLoss, 'GeM': GeM,\n                    'WarmupCosineDecay': WarmupCosineDecay}\n)\n\n# ✅ val_imgs_arr و val_labels اتبنوا في الـ Incremental Loop\nprint(f\"✅ Val images available: {len(val_imgs_arr):,}\")\nprint(f\"   Cancer  : {(val_labels == 1).sum():,}\")\nprint(f\"   Healthy : {(val_labels == 0).sum():,}\")\n\n\ndef tta_predict(model, imgs_uint8, n_tta=7):\n    \"\"\"\n    ✅ v10 FIXED:\n    - imgs_uint8: raw uint8 [0,255] — preprocess_input يتعمل جوه هنا بس\n    - TTA augmentations على الـ raw pixel values قبل preprocess\n    - مفيش double preprocessing\n    \"\"\"\n    all_preds = []\n    for tta_i in range(n_tta):\n        logits_this = []\n        for start in range(0, len(imgs_uint8), BATCH):\n            # ✅ copy ضروري عشان نعدل من غير ما نأثر على الأصل\n            batch = imgs_uint8[start:start+BATCH].copy().astype(np.float32)\n\n            # ── TTA augmentations — على raw float بعد التحويل ──\n            if tta_i == 1:\n                batch = batch[:, :, ::-1, :]     # horizontal flip\n            elif tta_i == 2:\n                batch = batch[:, ::-1, :, :]     # vertical flip\n            elif tta_i == 3:\n                batch = batch[:, :, ::-1, :]     # H + V flip\n                batch = batch[:, ::-1, :, :]\n            elif tta_i == 4:\n                # 90-degree rotation\n                batch = np.rot90(batch, k=1, axes=(1, 2))\n            elif tta_i == 5:\n                # 270-degree rotation\n                batch = np.rot90(batch, k=3, axes=(1, 2))\n            elif tta_i == 6:\n                # slight brightness shift\n                batch = np.clip(batch * 1.05, 0, 255)\n\n            # ✅ preprocess_input مرة واحدة بس هنا\n            batch = tf.keras.applications.efficientnet.preprocess_input(batch)\n\n            out = model(tf.constant(batch, dtype=tf.float32),\n                        training=False).numpy().flatten()\n            logits_this.extend(out)\n\n        probs = tf.sigmoid(tf.constant(logits_this, dtype=tf.float32)).numpy()\n        all_preds.append(probs)\n        print(f\"  TTA pass {tta_i+1}/{n_tta} — mean prob: {probs.mean():.4f}\")\n\n    return np.mean(all_preds, axis=0)\n\n\nprint(\"\\nRunning TTA inference (7 passes)...\")\n# ✅ val_imgs_arr هو uint8 raw — مش preprocessed\nval_resized = np.array([\n    cv2.resize(img, (IMG_SIZE, IMG_SIZE)) for img in val_imgs_arr\n], dtype=np.uint8)\n\nall_probs  = tta_predict(model, val_resized, n_tta=7)\nall_labels = val_labels.astype(int)\n\nif len(np.unique(all_labels)) < 2:\n    print(\"⚠️ Only one class in val — AUC can't be computed. Check your val set!\")\n    auc_score = 0.0\nelse:\n    auc_score = roc_auc_score(all_labels, all_probs)\n    print(f\"\\n✅ TTA AUC: {auc_score:.4f}\")\n\n# ── Threshold Sweep ───────────────────────────────────────────\ntest_thresholds = np.arange(0.05, 0.95, 0.01)\nf1_scores = []\nfor t in test_thresholds:\n    preds = (all_probs >= t).astype(int)\n    tp = ((preds == 1) & (all_labels == 1)).sum()\n    fp = ((preds == 1) & (all_labels == 0)).sum()\n    fn = ((preds == 0) & (all_labels == 1)).sum()\n    pr = tp / max(tp + fp, 1)\n    rc = tp / max(tp + fn, 1)\n    f1_scores.append(2 * pr * rc / max(pr + rc, 1e-6))\n\nbest_idx       = int(np.argmax(f1_scores))\nBEST_THRESHOLD = float(test_thresholds[best_idx])\nbest_f1        = f1_scores[best_idx]\ndefault_f1     = f1_scores[int(np.argmin(abs(test_thresholds - 0.5)))]\n\nprint(f\"ROC-AUC (TTA)  : {auc_score:.4f}\")\nprint(f\"Best Threshold : {BEST_THRESHOLD:.2f} → Cancer F1 = {best_f1:.4f}\")\nprint(f\"Default (0.50) :       → Cancer F1 = {default_f1:.4f}\")\n\n# ── Plots ─────────────────────────────────────────────────────\nif auc_score > 0:\n    fpr, tpr, _ = roc_curve(all_labels, all_probs)\n    pr_p, pr_r, _ = precision_recall_curve(all_labels, all_probs)\n\n    fig, axes = plt.subplots(1, 3, figsize=(18, 5))\n\n    axes[0].plot(fpr, tpr, color='#3B8BD4', lw=2, label=f'AUC = {auc_score:.3f}')\n    axes[0].plot([0,1],[0,1],'k--', alpha=0.5)\n    axes[0].set_xlabel('FPR'); axes[0].set_ylabel('TPR')\n    axes[0].set_title('ROC Curve (TTA)'); axes[0].legend()\n\n    axes[1].plot(pr_r, pr_p, color='#E8593C', lw=2)\n    axes[1].set_xlabel('Recall'); axes[1].set_ylabel('Precision')\n    axes[1].set_title('PR Curve (TTA)')\n\n    axes[2].plot(test_thresholds, f1_scores, color='#27ae60', lw=2)\n    axes[2].axvline(BEST_THRESHOLD, color='red', ls='--', label=f'Best={BEST_THRESHOLD:.2f}')\n    axes[2].axvline(0.5, color='gray', ls='--', alpha=0.5, label='Default=0.50')\n    axes[2].set_xlabel('Threshold'); axes[2].set_ylabel('Cancer F1')\n    axes[2].set_title('F1 vs Threshold'); axes[2].legend()\n\n    plt.tight_layout()\n    plt.savefig(os.path.join(SAVE_DIR, \"threshold_analysis_v10.png\"), dpi=150)\n    plt.show()\nprint(\"✅ Done!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:16:46.535817Z","iopub.status.idle":"2026-05-10T19:16:46.536494Z","shell.execute_reply.started":"2026-05-10T19:16:46.536358Z","shell.execute_reply":"2026-05-10T19:16:46.53638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ════════════════════════════════════════════════════════════════\n# CELL 15 — Final Evaluation بالـ Best Threshold + TTA\n# ════════════════════════════════════════════════════════════════\npreds_default = (all_probs >= 0.50).astype(int)\npreds_best    = (all_probs >= BEST_THRESHOLD).astype(int)\n\nprint(\"=\" * 60)\nprint(\"Results with DEFAULT threshold (0.50):\")\nprint(\"=\" * 60)\nprint(classification_report(all_labels, preds_default,\n                             target_names=[\"Healthy\", \"Cancer\"]))\n\nprint(\"=\" * 60)\nprint(f\"Results with BEST threshold ({BEST_THRESHOLD:.2f}) + TTA:\")\nprint(\"=\" * 60)\nprint(classification_report(all_labels, preds_best,\n                             target_names=[\"Healthy\", \"Cancer\"]))\n\n# ── Accuracy خصيصاً ──────────────────────────────────────────\nacc_default = (preds_default == all_labels).mean()\nacc_best    = (preds_best    == all_labels).mean()\nprint(f\"\\n🎯 Accuracy (default 0.50) : {acc_default:.4f}  ({acc_default*100:.2f}%)\")\nprint(f\"🎯 Accuracy (best thresh)  : {acc_best:.4f}  ({acc_best*100:.2f}%)\")\nprint(f\"🎯 ROC-AUC (TTA)           : {auc_score:.4f}\")\n\n# ── Confusion Matrices ────────────────────────────────────────\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 5))\nfor ax, preds, title in [\n    (ax1, preds_default, \"Threshold = 0.50 (Default)\"),\n    (ax2, preds_best,    f\"Threshold = {BEST_THRESHOLD:.2f} (Best F1) + TTA\")\n]:\n    cm = confusion_matrix(all_labels, preds)\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', ax=ax,\n                xticklabels=['Healthy','Cancer'],\n                yticklabels=['Healthy','Cancer'])\n    ax.set_title(title, fontsize=12)\n    ax.set_ylabel('Actual'); ax.set_xlabel('Predicted')\n\nplt.tight_layout()\nplt.savefig(os.path.join(SAVE_DIR, \"confusion_matrices_tta.png\"), dpi=150)\nplt.show()\nprint(f\"\\n✅ Best threshold ({BEST_THRESHOLD:.2f}) + TTA — use in production!\")\n","metadata":{"id":"EiCpnUaf2MuU","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:16:46.537444Z","iopub.status.idle":"2026-05-10T19:16:46.537798Z","shell.execute_reply.started":"2026-05-10T19:16:46.537619Z","shell.execute_reply":"2026-05-10T19:16:46.537641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ════════════════════════════════════════════════════════════════\n# CELL 16 — Predict على صور من الـ Val Set  ✅ v10\n# ════════════════════════════════════════════════════════════════\nimport pydicom\n\ndef predict_image(model, image_path: str, laterality: str = 'L',\n                  threshold: float = BEST_THRESHOLD) -> dict:\n    img = preprocess_image_cv2(image_path, laterality)\n    if img is None:\n        return {'error': 'Image not found or corrupted'}\n    img   = cv2.resize(img, (IMG_SIZE, IMG_SIZE)).astype(np.float32)\n    # ✅ preprocess_input مرة واحدة بس\n    img   = tf.keras.applications.efficientnet.preprocess_input(img)\n    img   = tf.expand_dims(img, 0)\n    logit = model(img, training=False).numpy()[0][0]\n    prob  = float(tf.sigmoid(tf.constant(logit, dtype=tf.float32)).numpy())\n    return {\n        'label'          : 'Cancer' if prob >= threshold else 'Healthy',\n        'probability'    : round(prob, 4),\n        'threshold_used' : threshold\n    }\n\n\n# ✅ df_val اتبنى في الـ Incremental Loop\n# ✅ بنعرض صور من الـ val set الحقيقي\nif 'image_path' not in df_val.columns:\n    df_val['image_path'] = df_val.apply(\n        lambda r: os.path.join(PATH_DATASET, 'train_images',\n                               str(r['patient_id']),\n                               str(r['image_id']) + '.dcm'), axis=1\n    )\n\n# عرض 2 cancer + 2 healthy من الـ val set\n# ✅ لو df_val مش فيه image paths حقيقية — بنعرض من val_imgs_arr مباشرة\nn_cancer_val  = (val_labels == 1).sum()\nn_healthy_val = (val_labels == 0).sum()\n\ncancer_idxs  = np.where(val_labels == 1)[0]\nhealthy_idxs = np.where(val_labels == 0)[0]\n\nshow_idxs = []\nif len(cancer_idxs)  >= 2: show_idxs += list(np.random.choice(cancer_idxs,  2, replace=False))\nelif len(cancer_idxs) > 0: show_idxs += list(cancer_idxs)\nif len(healthy_idxs) >= 2: show_idxs += list(np.random.choice(healthy_idxs, 2, replace=False))\nelif len(healthy_idxs) > 0: show_idxs += list(healthy_idxs)\n\nfig, axes = plt.subplots(1, len(show_idxs), figsize=(5*len(show_idxs), 6))\nif len(show_idxs) == 1: axes = [axes]\n\nfor ax, idx in zip(axes, show_idxs):\n    img_show = val_imgs_arr[idx]\n    actual   = 'Cancer' if val_labels[idx] == 1 else 'Healthy'\n\n    # ── Inference ──────────────────────────────────────────────\n    img_f = cv2.resize(img_show, (IMG_SIZE, IMG_SIZE)).astype(np.float32)\n    img_f = tf.keras.applications.efficientnet.preprocess_input(img_f)\n    img_t = tf.expand_dims(img_f, 0)\n    logit = model(img_t, training=False).numpy()[0][0]\n    prob  = float(tf.sigmoid(tf.constant(logit, dtype=tf.float32)).numpy())\n    pred  = 'Cancer' if prob >= BEST_THRESHOLD else 'Healthy'\n\n    ax.imshow(img_show[:,:,0], cmap='gray')\n    color = 'green' if actual == pred else 'red'\n    ax.set_title(\n        f\"Actual : {actual}\\nPred   : {pred}\\nProb   : {prob:.1%}\",\n        color=color, fontsize=11\n    )\n    ax.axis('off')\n\nplt.suptitle(f'Predictions  |  Threshold = {BEST_THRESHOLD:.2f}', fontsize=13)\nplt.tight_layout()\nplt.show()\n\nprint(f'\\n✅ All files saved to: {SAVE_DIR}')\nprint(f'   - best_model.keras')\nprint(f'   - training_history.png')\nprint(f'   - threshold_analysis_v10.png')\nprint(f'   - confusion_matrices.png')\nprint(f'   - training_log.csv')\nprint(f'   - incremental_cumulative_val.png')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:16:46.538921Z","iopub.status.idle":"2026-05-10T19:16:46.539283Z","shell.execute_reply.started":"2026-05-10T19:16:46.539111Z","shell.execute_reply":"2026-05-10T19:16:46.539143Z"}},"outputs":[],"execution_count":null}]}