{"nbformat":4,"nbformat_minor":4,"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.0"}},"cells":[{"cell_type":"code","execution_count":null,"metadata":{},"outputs":[],"source":"\"\"\"\nRSNA Knee Abnormality Detection â€” Working Baseline\n==================================================\n\nSingle-file, Kaggle-ready baseline for:\nhttps://www.kaggle.com/competitions/rsna-knee-abnormality-detection\n\nApproach (deliberately simple, submits reliably):\n  1. For each series, sample a handful of representative slices.\n  2. From each slice extract robust pixel statistics (mean, std, percentiles,\n     histogram, edge density) + a few safe DICOM header fields.\n  3. Aggregate per-series features to one feature vector per study\n     (grouped by anatomical plane so plane-specific signal is preserved).\n  4. Train one HistGradientBoostingClassifier per label (multi-label).\n  5. Validate with BOTH a random StratifiedKFold on label-sum, AND a\n     scanner-grouped KFold (grouping by Manufacturer + model + software).\n     Report macro-averaged ROC AUC (the competition metric) for each.\n  6. Predict on the test set and write submission.csv matching\n     sample_submission.csv row-for-row.\n\nOnly sklearn / numpy / pandas / pydicom (all on Kaggle's default image, so\nthis runs with internet disabled). No pretrained weights required.\n\nThe script is organised as cell-style blocks (`# %% ...`) so it can be pasted\ndirectly into a Kaggle notebook.\n\"\"\"\n\n# %% [imports] --------------------------------------------------------------\nfrom __future__ import annotations\n\nimport os\nimport gc\nimport re\nimport sys\nimport math\nimport time\nimport warnings\nfrom pathlib import Path\nfrom typing import Iterable\n\nimport numpy as np\nimport pandas as pd\n\nwarnings.filterwarnings(\"ignore\")\n\n# %% [config] ---------------------------------------------------------------\n# Kaggle mounts competition data in one of two layouts depending on how the\n# notebook was attached:\n#   /kaggle/input/<slug>/                       (dataset-style attach)\n#   /kaggle/input/competitions/<slug>/          (competition-style attach)\n# We probe both and pick whichever actually contains the CSVs, so the script\n# works regardless of how the user added the dataset. Override with the\n# RSNA_DATA_DIR environment variable if running elsewhere.\ndef _locate_data_dir() -> Path:\n    env = os.environ.get(\"RSNA_DATA_DIR\")\n    if env:\n        return Path(env)\n    candidates: list[Path] = []\n    slug = \"rsna-knee-abnormality-detection\"\n    for base in (\"/kaggle/input\", \"/kaggle/input/competitions\"):\n        candidates.append(Path(base) / slug)\n    # As a last resort scan /kaggle/input for any dir with train.csv +\n    # sample_submission.csv â€” covers renamed attachments.\n    root = Path(\"/kaggle/input\")\n    if root.is_dir():\n        for p in root.rglob(\"sample_submission.csv\"):\n            candidates.append(p.parent)\n    for c in candidates:\n        if (c / \"train.csv\").is_file() and (c / \"sample_submission.csv\").is_file():\n            return c\n    # Fallback: return the first candidate so error messages are informative.\n    return candidates[0] if candidates else Path(\"/kaggle/input\")\n\n\nDATA_DIR = _locate_data_dir()\nOUT_DIR = Path(os.environ.get(\"RSNA_OUT_DIR\", \"/kaggle/working\"))\nOUT_DIR.mkdir(parents=True, exist_ok=True)\n\n# Twelve target columns, in the exact order required by the submission header.\nLABEL_COLS: list[str] = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\",\n    \"Medial OA\", \"Lateral OA\", \"PF OA\",\n    \"Effusion\", \"Synovitis\", \"Baker's\",\n    \"Contusion\", \"Fracture\",\n]\n\n# Sampling knobs. Keep low so the baseline fits comfortably in the 9-hour\n# CPU budget on ~1,300 test studies; scale up on a stronger baseline.\nSLICES_PER_SERIES = 3           # 3â†’5 more stable, but 5 slices breaks 9h CPU budget\nMAX_SERIES_PER_STUDY = 8        # was 6; median-series-per-study on train is 5\nUSE_25D_INPUT = True            # 3 adjacent slices as R/G/B channels (cheap +0.02-0.04 AUC)\nN_HIST_BINS = 8                 # coarse intensity histogram\nN_FOLDS = 3                     # 5â†’3 saves 40% CV time; keeps us inside 9h CPU budget\nRANDOM_STATE = 42\nPLANES = (\"Sagittal\", \"Coronal\", \"Axial\")\n# v5 showed 0.79 random vs 0.64 grouped AUC â†’ the meta_* DICOM header\n# features (TR/TE/thickness/spacing/field strength/imaging frequency) leak\n# scanner identity. Prevalence differs across public/private test per the\n# brief, so this shortcut wins public LB and loses private. Keep off.\nUSE_SCANNER_METADATA_FEATURES = False\n# v14 bisect: v13 (advanced parser) dropped grouped AUC 0.700 â†’ 0.688 despite\n# recovering 1,249 extra positive rows. Toggle to isolate whether the parser\n# upgrade OR the DINOv2/windowing/ordering stack is the regressor. False = use\n# v10's binary regex parser + flat sample_weight. True = v13's soft parser + conf.\nUSE_ADVANCED_PARSER = False\n# v15/v16 DISABLED: end-to-end DINOv2 SlotHead fine-tune. On v5-Kaggle it\n# collapsed to per-label priors (holdout AUC = 0.5000, submission had 3\n# identical rows) â€” head too small, LR_BACKBONE too low, 3 epochs too few\n# for image signal to overwrite the class-prior floor. v6 reverts to the\n# frozen-DINOv2 â†’ HistGradientBoosting branch below (much stronger data-\n# efficient signal for this dataset size). Re-enable only after training-\n# regime fixes (unfreeze more, higher LR, â‰¥10 epochs).\n# v15: end-to-end DINOv2 SlotHead fine-tune (see end2end.py). Replaces the\n# HistGradientBoosting frozen-feature path entirely. On any failure, the\n# __main__ try/except falls back to writing sample_submission.\nUSE_END2END = False\n# Holdout fraction for end-to-end AUC measurement (no k-fold â€” too expensive).\nE2E_HOLDOUT = 0.20\n# v7: Protocol-count features (n_series, n_series_<plane>, n_fluid_sensitive,\n# n_fat_suppression) are scanner-protocol IDs â€” different sites use different\n# plane counts and sequence protocols. v6 CV showed random-fold 0.837 vs\n# grouped-fold 0.706 (Î”=0.131) â€” HGB was using these to memorise scanner\n# identity, inflating random-fold and hurting LB. Off by default in v7.\nUSE_PROTOCOL_COUNT_FEATURES = False\n# v7: Ensemble HGB + LogisticRegression per label (60/40 weighted mean).\n# LR generalises better under scanner shift because trees over-memorise\n# feature-value combinations. Small cost, aim: +0.02-0.04 macro AUC.\nUSE_LR_ENSEMBLE = True\n# v7: Offline LLM-based report parser. Requires a multilingual sentence-\n# transformer attached as a Kaggle dataset (e.g.\n# `paraphrase-multilingual-MiniLM-L12-v2`). Falls back to the v10 regex\n# parser if none is found. LLM path uses report â†’ embedding, cosine sim\n# to label-prototype embeddings, threshold â†’ 0/1. Cleaner labels for\n# multilingual reports, aim: +0.03-0.06 macro AUC on the LB.\nUSE_LLM_PARSER = True\nLLM_SIM_THRESHOLD = 0.35\n\n# Optional: subsample training studies for iteration speed. Set to None to\n# use all training studies. On the real submission run leave it None.\nTRAIN_SUBSAMPLE = int(os.environ.get(\"RSNA_TRAIN_SUBSAMPLE\", \"0\")) or None\n\n# %% [report â†’ weak labels] -------------------------------------------------\n# train.csv ships with labels populated ONLY for the ~58 expert-annotated\n# studies; the other ~4349 rows have NaN targets. If we fillna(0) we lie to\n# the model on 98 % of the training set â€” that's why the vanilla baseline\n# hovers at 0.56 AUC. Report parsing recovers real signal for those rows.\n#\n# Reports may be in English / Spanish / French / German â€” we include common\n# terms for each. Simple sentence-level negation (\"no ACL tear\", \"sin lesiÃ³n\")\n# flips a match to 0.\n\n_LABEL_KEYWORDS: dict[str, list[str]] = {\n    \"ACL\": [r\"\\bACL\\b\", r\"anterior cruciate\", r\"cruzado anterior\",\n            r\"vorderes kreuzband\", r\"ligament croisÃ© antÃ©rieur\"],\n    \"MCL\": [r\"\\bMCL\\b\", r\"medial collateral\", r\"colateral medial\",\n            r\"innenband\", r\"ligament collatÃ©ral mÃ©dial\"],\n    \"Medial Meniscus\":  [r\"medial meniscus\", r\"menisco medial\", r\"innenmeniskus\",\n                         r\"mÃ©nisque mÃ©dial\", r\"mÃ©nisque interne\"],\n    \"Lateral Meniscus\": [r\"lateral meniscus\", r\"menisco lateral\", r\"auÃŸenmeniskus\",\n                         r\"mÃ©nisque latÃ©ral\", r\"mÃ©nisque externe\"],\n    # OA labels: radiologists rarely write \"arthritis\" â€” much more often\n    # \"degenerative\", \"joint space narrowing\", \"chondromalacia\", \"cartilage\n    # loss/thinning\", \"osteophytes/spurring\". Per-compartment, all four langs.\n    \"Medial OA\":  [\n        r\"medial (compartment )?(osteo)?arthr\",\n        r\"medial (compartment )?(joint space narrow|jsn)\",\n        r\"medial (compartment )?(cartilage (loss|thinning|defect)|chondr(al|omalacia|osis|opathy))\",\n        r\"medial (compartment )?(degenerat|osteophyt|spurring|bone[- ]on[- ]bone)\",\n        r\"medial (femorotibial|tibiofemoral).{0,30}(arthr|degenerat|narrow|chondr)\",\n        r\"artrosis medial\", r\"gonartrosis medial\", r\"condropatÃ­a medial\",\n        r\"mediale gonarthrose\", r\"innere gonarthrose\", r\"mediale chondropathie\",\n        r\"arthrose (fÃ©moro[- ]?tibiale mÃ©diale|mÃ©diale)\", r\"chondropathie mÃ©diale\",\n    ],\n    \"Lateral OA\": [\n        r\"lateral (compartment )?(osteo)?arthr\",\n        r\"lateral (compartment )?(joint space narrow|jsn)\",\n        r\"lateral (compartment )?(cartilage (loss|thinning|defect)|chondr(al|omalacia|osis|opathy))\",\n        r\"lateral (compartment )?(degenerat|osteophyt|spurring|bone[- ]on[- ]bone)\",\n        r\"lateral (femorotibial|tibiofemoral).{0,30}(arthr|degenerat|narrow|chondr)\",\n        r\"artrosis lateral\", r\"gonartrosis lateral\", r\"condropatÃ­a lateral\",\n        r\"laterale gonarthrose\", r\"Ã¤uÃŸere gonarthrose\", r\"laterale chondropathie\",\n        r\"arthrose (fÃ©moro[- ]?tibiale latÃ©rale|latÃ©rale)\", r\"chondropathie latÃ©rale\",\n    ],\n    \"PF OA\": [\n        r\"patellofemoral (osteo)?arthr\",\n        r\"patellofemoral (joint space narrow|jsn)\",\n        r\"patellofemoral (cartilage (loss|thinning|defect)|chondr(al|omalacia|osis|opathy))\",\n        r\"patellofemoral (degenerat|osteophyt|spurring)\",\n        r\"chondromalacia patellae?\", r\"patellar chondr\",\n        r\"trochlear (cartilage|chondr)\", r\"retropatellar (chondr|cartilage|arthr|degenerat)\",\n        r\"artrosis (patelo|patelofemoral|femoropatelar)\",\n        r\"condropatÃ­a (patelo|patelofemoral|rotuliana)\",\n        r\"retropatellare arthrose\", r\"femoropatellare (arthrose|chondropathie)\",\n        r\"arthrose fÃ©moro[- ]?patellaire\", r\"chondropathie rotulienne\",\n    ],\n    \"Effusion\":  [r\"effusion\", r\"joint fluid\", r\"derrame\", r\"erguss\", r\"Ã©panchement\"],\n    \"Synovitis\": [r\"synoviti\", r\"sinoviti\", r\"synovialiti\", r\"synovitis\"],\n    \"Baker's\":   [r\"baker'?s? cyst\", r\"popliteal cyst\", r\"quiste de baker\",\n                  r\"bakerzyste\", r\"kyste de baker\", r\"kyste poplitÃ©\"],\n    \"Contusion\": [r\"contusion\", r\"bone bruise\", r\"bone marrow (o?edema|edema)\",\n                  r\"contusiÃ³n\", r\"knochenmarksÃ¶dem\", r\"contusion osseuse\"],\n    \"Fracture\":  [r\"fractur\", r\"fisura Ã³sea\", r\"knochenbruch\", r\"fraktur\",\n                  r\"break in the (cortex|bone)\"],\n}\n\n_NEG_TOKENS = (\n    r\"\\bno\\b\", r\"\\bwithout\\b\", r\"\\babsence of\\b\", r\"\\bintact\\b\", r\"\\bnegative for\\b\",\n    r\"\\bsin\\b\", r\"\\bausencia de\\b\", r\"\\bkein\\b\", r\"\\bkeine\\b\", r\"\\bohne\\b\",\n    r\"\\bpas de\\b\", r\"\\bsans\\b\", r\"\\baucun\\b\", r\"\\bnegatif\\b\",\n)\n_NEG_RE = re.compile(\"|\".join(_NEG_TOKENS), re.IGNORECASE)\n\n\ndef _labels_from_report(text) -> dict[str, float]:\n    \"\"\"Return {label -> 1.0 | 0.0 | nan} from one free-text report.\"\"\"\n    out: dict[str, float] = {c: float(\"nan\") for c in LABEL_COLS}\n    if not isinstance(text, str) or not text.strip():\n        return out\n    # Split into rough sentences so negation only fires within a local window.\n    sentences = re.split(r\"[.\\n;Â·â€¢]+\", text)\n    for sent in sentences:\n        s = sent.strip()\n        if not s:\n            continue\n        negated = bool(_NEG_RE.search(s))\n        for lbl, patterns in _LABEL_KEYWORDS.items():\n            for p in patterns:\n                if re.search(p, s, re.IGNORECASE):\n                    val = 0.0 if negated else 1.0\n                    prev = out[lbl]\n                    # Positive evidence wins over negative and over NaN.\n                    if math.isnan(prev) or (prev == 0.0 and val == 1.0):\n                        out[lbl] = val\n                    break\n    return out\n\n\n# --- v7: offline LLM parser ------------------------------------------------\n# Uses a multilingual sentence-transformer to embed each Report, then scores\n# per-label similarity against a set of pre-encoded prototype phrases. No\n# generation â†’ cheap (batched embeddings on GPU, ~5 min for 4400 reports).\n# Falls back silently to the v10 regex parser if no model is attached.\n\n_LLM: dict = {\n    \"encoder\": None, \"tokenizer\": None, \"device\": None,\n    \"prototypes\": None, \"proto_ranges\": None, \"mode\": \"none\",\n}\n\n# 4-7 English phrasings per label â€” cross-lingual embeddings map paraphrases\n# in Spanish/French/German/etc into the same latent region.\n_LLM_LABEL_PROTOTYPES: dict[str, list[str]] = {\n    \"ACL\": [\"anterior cruciate ligament tear\", \"ACL tear\", \"ACL rupture\",\n            \"torn anterior cruciate ligament\", \"ACL is disrupted\",\n            \"complete tear of the anterior cruciate ligament\"],\n    \"MCL\": [\"medial collateral ligament tear\", \"MCL sprain\",\n            \"MCL is torn\", \"medial collateral ligament injury\",\n            \"grade II MCL sprain\"],\n    \"Medial Meniscus\":  [\"medial meniscus tear\", \"tear of the medial meniscus\",\n                         \"medial meniscal tear\", \"horizontal tear medial meniscus\",\n                         \"bucket-handle tear medial meniscus\"],\n    \"Lateral Meniscus\": [\"lateral meniscus tear\", \"tear of the lateral meniscus\",\n                         \"lateral meniscal tear\", \"radial tear lateral meniscus\"],\n    \"Medial OA\": [\"medial compartment osteoarthritis\",\n                  \"medial joint space narrowing\",\n                  \"medial compartment chondral loss\",\n                  \"osteoarthritis of the medial tibiofemoral compartment\"],\n    \"Lateral OA\": [\"lateral compartment osteoarthritis\",\n                   \"lateral joint space narrowing\",\n                   \"lateral compartment chondral loss\",\n                   \"osteoarthritis of the lateral tibiofemoral compartment\"],\n    \"PF OA\": [\"patellofemoral osteoarthritis\", \"patellofemoral chondromalacia\",\n              \"patellofemoral compartment osteoarthritis\",\n              \"retropatellar chondral loss\"],\n    \"Effusion\":  [\"joint effusion\", \"knee effusion\", \"large joint effusion\",\n                  \"moderate suprapatellar effusion\", \"fluid in the joint\"],\n    \"Synovitis\": [\"synovitis\", \"synovial thickening\",\n                  \"synovial inflammation\", \"reactive synovitis\"],\n    \"Baker's\":   [\"Baker's cyst\", \"popliteal cyst\",\n                  \"large Baker's cyst in the popliteal fossa\",\n                  \"ruptured Baker's cyst\"],\n    \"Contusion\": [\"bone contusion\", \"bone bruise\", \"bone marrow edema\",\n                  \"trabecular oedema\", \"post-traumatic bone marrow oedema\"],\n    \"Fracture\":  [\"fracture\", \"cortical fracture\", \"trabecular fracture\",\n                  \"avulsion fracture\", \"impaction fracture\"],\n}\n\n\ndef _mean_pool_embeddings(hidden, mask):\n    \"\"\"Standard sentence-transformer mean pooling with attention mask.\"\"\"\n    m = mask.unsqueeze(-1).float()\n    summed = (hidden * m).sum(dim=1)\n    counts = m.sum(dim=1).clamp(min=1e-6)\n    return summed / counts\n\n\ndef _l2_normalize_torch(x, eps=1e-6):\n    return x / x.norm(dim=1, keepdim=True).clamp(min=eps)\n\n\ndef _try_setup_llm_parser() -> bool:\n    \"\"\"Try to load a multilingual sentence-transformer from /kaggle/input.\n    Returns True on success. Silent no-op on any failure â€” caller falls\n    back to the regex parser.\"\"\"\n    if not USE_LLM_PARSER:\n        return False\n    try:\n        import torch\n        from transformers import AutoTokenizer, AutoModel\n    except Exception as e:\n        print(f\"[llm] transformers unavailable ({e}); using regex parser.\")\n        return False\n\n    # Search under /kaggle/input for a directory containing config.json AND\n    # tokenizer.json / vocab.txt / spiece.model (any HF model layout).\n    root = Path(\"/kaggle/input\")\n    if not root.is_dir():\n        return False\n    hits: list[Path] = []\n    prefer = (\"multilingual\", \"xlm\", \"mpnet\", \"labse\", \"distiluse\", \"minilm\")\n    for cfg in root.rglob(\"config.json\"):\n        d = cfg.parent\n        # Basic sanity: has at least one weights file.\n        if not any(d.glob(\"*.bin\")) and not any(d.glob(\"*.safetensors\")):\n            continue\n        hits.append(d)\n    if not hits:\n        print(\"[llm] no HF model dir found under /kaggle/input; using regex parser. \"\n              \"Attach `sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2` \"\n              \"(or similar) to enable the LLM path.\")\n        return False\n    # Prefer multilingual sentence-transformer style directories.\n    hits.sort(key=lambda p: (\n        0 if any(k in p.name.lower() for k in prefer) else 1,\n        len(p.parts),\n    ))\n    model_dir = hits[0]\n\n    try:\n        tokenizer = AutoTokenizer.from_pretrained(str(model_dir))\n        model = AutoModel.from_pretrained(str(model_dir))\n        model.eval()\n        device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n        model = model.to(device)\n    except Exception as e:\n        print(f\"[llm] failed to load model from {model_dir} ({e}); using regex parser.\")\n        return False\n\n    # Pre-encode all label prototypes once. Shape: (P, D).\n    proto_texts: list[str] = []\n    proto_ranges: dict[str, tuple[int, int]] = {}\n    idx = 0\n    for lbl in LABEL_COLS:\n        phrases = _LLM_LABEL_PROTOTYPES.get(lbl, [lbl.lower()])\n        proto_ranges[lbl] = (idx, idx + len(phrases))\n        proto_texts.extend(phrases)\n        idx += len(phrases)\n    try:\n        with torch.no_grad():\n            enc = tokenizer(proto_texts, padding=True, truncation=True,\n                            max_length=48, return_tensors=\"pt\").to(device)\n            out = model(**enc)\n            proto = _mean_pool_embeddings(out.last_hidden_state,\n                                          enc[\"attention_mask\"])\n            proto = _l2_normalize_torch(proto)\n    except Exception as e:\n        print(f\"[llm] failed to encode prototypes ({e}); using regex parser.\")\n        return False\n\n    _LLM[\"encoder\"] = model\n    _LLM[\"tokenizer\"] = tokenizer\n    _LLM[\"device\"] = device\n    _LLM[\"prototypes\"] = proto\n    _LLM[\"proto_ranges\"] = proto_ranges\n    _LLM[\"mode\"] = f\"llm:{model_dir.name}\"\n    print(f\"[llm] enabled: {_LLM['mode']} @ {device}, {len(proto_texts)} prototypes\")\n    return True\n\n\ndef _llm_score_reports(texts: list[str]) -> np.ndarray:\n    \"\"\"Batch score N reports Ã— 12 labels. Returns (N, 12) max-cosine-sim.\"\"\"\n    import torch\n    tokenizer = _LLM[\"tokenizer\"]\n    model = _LLM[\"encoder\"]\n    device = _LLM[\"device\"]\n    protos = _LLM[\"prototypes\"]           # (P, D) L2-normalized\n    ranges = _LLM[\"proto_ranges\"]\n\n    out = np.zeros((len(texts), len(LABEL_COLS)), dtype=np.float32)\n    B = 32\n    for i in range(0, len(texts), B):\n        batch = texts[i:i + B]\n        with torch.no_grad():\n            enc = tokenizer(batch, padding=True, truncation=True,\n                            max_length=384, return_tensors=\"pt\").to(device)\n            hs = model(**enc).last_hidden_state\n            emb = _mean_pool_embeddings(hs, enc[\"attention_mask\"])\n            emb = _l2_normalize_torch(emb)         # (b, D)\n            sims = emb @ protos.T                  # (b, P)\n        sims_np = sims.cpu().numpy()\n        for j, lbl in enumerate(LABEL_COLS):\n            lo, hi = ranges[lbl]\n            out[i:i + len(batch), j] = sims_np[:, lo:hi].max(axis=1)\n    return out\n\n\ndef _merge_expert_and_report_labels(train_csv: pd.DataFrame) -> tuple[pd.DataFrame, pd.Series, pd.DataFrame]:\n    \"\"\"\n    Build the true label matrix by combining expert labels (rows where the CSV\n    already has non-NaN values) with report-derived labels for the rest.\n\n    Report labels come from the ported multilingual parser (report_parser.extract)\n    as (score, confidence) pairs. Score > 0.5 â†’ positive; confidence is\n    returned per row/label so it can be piped into sample_weight during\n    training (weak/low-conf rows pull less than strong-conf ones).\n\n    Returns (labels_df, is_expert bool Series, conf_df).\n    \"\"\"\n    df = train_csv.set_index(\"StudyInstanceUID\")\n    labels = df[LABEL_COLS].copy()\n    is_expert = labels.notna().any(axis=1)\n    # Expert rows always full confidence; weak rows get 1.0 in the binary\n    # parser branch (no per-row confidence) or the parser's own conf in the\n    # advanced branch.\n    conf = pd.DataFrame(1.0, index=labels.index, columns=LABEL_COLS)\n\n    if USE_ADVANCED_PARSER:\n        # Lazy import â€” the notebook builder concatenates report_parser into the\n        # same cell before this file, so the name resolves at runtime.\n        try:\n            from report_parser import extract as _rp_extract  # local file\n        except Exception:\n            _rp_extract = extract  # type: ignore[name-defined]  # from concatenated cell\n    else:\n        _rp_extract = None  # binary regex path uses _labels_from_report\n\n    if \"Report\" in df.columns:\n        need = labels.index[~is_expert]\n        # LLM path: batch all weak-label reports through sentence-transformer.\n        if USE_LLM_PARSER and _LLM[\"mode\"] != \"none\":\n            texts = [str(df.loc[sid, \"Report\"]) for sid in need]\n            sim_scores = _llm_score_reports(texts)   # (N, 12) max-cosine-sim\n            for row_i, sid in enumerate(need):\n                for col_j, c in enumerate(LABEL_COLS):\n                    sim = float(sim_scores[row_i, col_j])\n                    labels.loc[sid, c] = int(sim >= LLM_SIM_THRESHOLD)\n                    conf.loc[sid, c] = min(1.0, sim / max(LLM_SIM_THRESHOLD, 1e-6))\n        else:\n            for sid in need:\n                if _rp_extract is None:\n                    # v10 path: binary regex parser, no per-label confidence.\n                    bin_labels = _labels_from_report(df.loc[sid, \"Report\"])\n                    for c in LABEL_COLS:\n                        v = bin_labels.get(c, float(\"nan\"))\n                        labels.loc[sid, c] = 0 if math.isnan(v) else int(v)\n                    continue\n                r = _rp_extract(df.loc[sid, \"Report\"])\n                for c in LABEL_COLS:\n                    labels.loc[sid, c] = 1 if r.get(c, 0.0) > 0.5 else 0\n                    conf.loc[sid, c] = float(r.get(c + \"__conf\", 0.05))\n\n    labels = labels.fillna(0).astype(int)\n    return labels, is_expert, conf\n\n\n# %% [dicom utilities] ------------------------------------------------------\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\n# pydicom needs pylibjpeg/gdcm for JPEG-Lossless / JPEG-2000 transfer\n# syntaxes. Both are preinstalled on Kaggle's Python image; we import them\n# lazily and swallow errors so the script still runs on a bare environment\n# (uncompressed / Implicit VR files will still decode).\nfor _mod in (\"pylibjpeg\", \"gdcm\"):\n    try:\n        __import__(_mod)\n    except Exception:\n        pass\n\n\ndef _safe_read_dcm(path: Path) -> pydicom.dataset.FileDataset | None:\n    try:\n        return pydicom.dcmread(str(path), force=True)\n    except Exception:\n        return None\n\n\ndef _pixels_or_none(ds) -> np.ndarray | None:\n    \"\"\"Return the raw calibrated float pixel array, or None on failure.\n\n    Applies DICOM RescaleSlope + RescaleIntercept and inverts MONOCHROME1\n    (where BRIGHT means LOW signal). Does NOT normalise intensity â€” the\n    caller must window per-series, not per-slice, otherwise a slice\n    through a large effusion is stretched onto the same range as a dry\n    bone slice and effusion/synovitis contrast is deleted (see\n    wguesdon/dinov2-at-meniscus-resolution cell 24).\n    \"\"\"\n    try:\n        arr = ds.pixel_array\n    except Exception:\n        return None\n    if arr is None or arr.size == 0:\n        return None\n    arr = np.asarray(arr, dtype=np.float32)\n    if arr.ndim == 3:\n        arr = arr[arr.shape[0] // 2]\n    if arr.ndim != 2:\n        return None\n    try:\n        slope = float(getattr(ds, \"RescaleSlope\", 1.0) or 1.0)\n        intercept = float(getattr(ds, \"RescaleIntercept\", 0.0) or 0.0)\n        arr = arr * slope + intercept\n        if str(getattr(ds, \"PhotometricInterpretation\", \"\")).upper() == \"MONOCHROME1\":\n            arr = float(np.nanmax(arr)) - arr\n    except Exception:\n        pass\n    if not np.isfinite(arr).any():\n        return None\n    return arr\n\n\ndef _series_window(arrays: list[np.ndarray], lo_pct=1.0, hi_pct=99.0) -> tuple[float, float]:\n    \"\"\"1st..99th percentile intensity window over pooled slices of ONE series.\n    Preserves relative brightness across slices (effusion still looks bright\n    vs bone), unlike per-slice min/max.\"\"\"\n    pool = []\n    for a in arrays:\n        flat = a[::3, ::3].ravel()\n        flat = flat[np.isfinite(flat)]\n        if flat.size:\n            pool.append(flat)\n    if not pool:\n        return 0.0, 1.0\n    pool = np.concatenate(pool)\n    lo, hi = np.percentile(pool, [lo_pct, hi_pct])\n    if not (np.isfinite(lo) and np.isfinite(hi)) or hi <= lo:\n        return float(pool.min()), float(pool.max()) if float(pool.max()) > float(pool.min()) else float(pool.min()) + 1.0\n    return float(lo), float(hi)\n\n\ndef _apply_window(arr: np.ndarray, lo: float, hi: float) -> np.ndarray:\n    \"\"\"Window an intensity array to [0, 1] with given lo/hi cutoffs.\"\"\"\n    return np.clip((arr - lo) / max(hi - lo, 1e-6), 0.0, 1.0).astype(np.float32)\n\n\ndef _series_hist_equalize(windowed: dict[int, np.ndarray], nbins: int = 256) -> dict[int, np.ndarray]:\n    \"\"\"Histogram-equalise all slices in a series using the pooled series CDF.\n    Maps the empirical intensity distribution to uniform [0,1] â€” removes\n    scanner-specific contrast bias while preserving within-series structure\n    (effusion remains brighter than bone because the CDF is monotone).\"\"\"\n    if not windowed:\n        return windowed\n    all_vals = np.concatenate([a.ravel() for a in windowed.values()])\n    counts, edges = np.histogram(all_vals, bins=nbins, range=(0.0, 1.0))\n    cdf = np.cumsum(counts).astype(np.float64)\n    cdf = (cdf - cdf[0]) / max(cdf[-1] - cdf[0], 1.0)\n    result: dict[int, np.ndarray] = {}\n    for i, a in windowed.items():\n        idx = np.searchsorted(edges[1:], a.ravel(), side='left')\n        idx = np.clip(idx, 0, nbins - 1)\n        result[i] = cdf[idx].reshape(a.shape).astype(np.float32)\n    return result\n\n\ndef _slice_features(img: np.ndarray) -> np.ndarray:\n    \"\"\"Compact per-slice descriptor: stats + histogram + edge density.\"\"\"\n    flat = img.ravel()\n    stats = [\n        float(flat.mean()),\n        float(flat.std()),\n        float(np.percentile(flat, 10)),\n        float(np.percentile(flat, 50)),\n        float(np.percentile(flat, 90)),\n        float((flat > 0.5).mean()),                 # bright-tissue fraction\n    ]\n    hist, _ = np.histogram(flat, bins=N_HIST_BINS, range=(0.0, 1.0))\n    hist = hist.astype(np.float32) / max(flat.size, 1)\n    # Cheap edge-density proxy: mean absolute gradient magnitude.\n    gy = np.abs(np.diff(img, axis=0)).mean() if img.shape[0] > 1 else 0.0\n    gx = np.abs(np.diff(img, axis=1)).mean() if img.shape[1] > 1 else 0.0\n    return np.concatenate([stats, hist, [gx, gy, img.shape[0], img.shape[1]]])\n\n\n# Names come from the per-slice descriptor above; keep in sync if changed.\n_PIXEL_STAT_NAMES = (\n    [\"mean\", \"std\", \"p10\", \"p50\", \"p90\", \"bright_frac\"]\n    + [f\"hist{i}\" for i in range(N_HIST_BINS)]\n    + [\"grad_x\", \"grad_y\", \"rows\", \"cols\"]\n)\n\n\n# %% [pretrained CNN feature extractor] -------------------------------------\n# ResNet18 penultimate features (512-dim) computed on GPU. These are far more\n# scanner-invariant than raw pixel statistics because ImageNet training\n# saw huge intensity/contrast diversity. We try to set it up at startup; if\n# torch / torchvision / the weights file are missing, we transparently fall\n# back to pixel statistics so the notebook still submits.\nclass _CNNState:\n    mode: str = \"pixel_stats\"          # \"cnn\" once activated\n    feature_size: int = len(_PIXEL_STAT_NAMES)\n    feature_names: list[str] = list(_PIXEL_STAT_NAMES)\n    model = None                        # torch.nn.Module\n    device = None                       # \"cuda\" or \"cpu\"\n    transform = None                    # callable(np.ndarray) -> torch.Tensor\n\n_CNN = _CNNState()\n\n\ndef _torch_load_safe(path, map_location=\"cpu\"):\n    \"\"\"torch.load with weights_only=False for trusted Kaggle-hosted pretrained\n    weights; PyTorch 2.6+ requires this for legacy .tar pickles. Old torch\n    versions don't take the kwarg â€” TypeError â†’ fall through.\"\"\"\n    import torch\n    try:\n        return torch.load(str(path), map_location=map_location, weights_only=False)\n    except TypeError:\n        return torch.load(str(path), map_location=map_location)\n\n\ndef _try_setup_dinov2(torch, transforms) -> bool:\n    \"\"\"Prefer DINOv2 (Meta self-supervised ViT â€” dominant medical-imaging\n    backbone in 2025-26 public solutions). Requires timm + a Kaggle-hosted\n    dinov2 weights .pth attached. Returns True if activated.\n    \"\"\"\n    try:\n        import timm\n    except Exception:\n        return False\n    # Search for any DINOv2 checkpoint under /kaggle/input.\n    weight_paths = list(Path(\"/kaggle/input\").rglob(\"dinov2*.pth\")) \\\n                 + list(Path(\"/kaggle/input\").rglob(\"dinov2*.bin\")) \\\n                 + list(Path(\"/kaggle/input\").rglob(\"dinov2*.safetensors\"))\n    if not weight_paths:\n        return False\n    # Prefer the smallest ViT to keep runtime realistic on T4.\n    weight_paths.sort(key=lambda p: (0 if \"vits14\" in p.name else\n                                     1 if \"vitb14\" in p.name else\n                                     2 if \"vitl14\" in p.name else 3))\n    weight_path = weight_paths[0]\n    # Map filename â†’ timm model name and feature dim.\n    name_lookup = [\n        (\"vits14\", \"vit_small_patch14_dinov2.lvd142m\", 384),\n        (\"vitb14\", \"vit_base_patch14_dinov2.lvd142m\", 768),\n        (\"vitl14\", \"vit_large_patch14_dinov2.lvd142m\", 1024),\n    ]\n    tag_model_dim = next(((t, m, d) for (t, m, d) in name_lookup\n                          if t in weight_path.name), None)\n    if tag_model_dim is None:\n        return False\n    tag, model_name, feat_dim = tag_model_dim\n    try:\n        # ViT-S/14 was pretrained at 518Ã—518. We want 224 for speed/memory\n        # (224/14 = 16 patches exactly). Passing img_size=224 makes timm\n        # rebuild the patch embed and interpolate the position embeddings\n        # instead of asserting the resolution mismatch.\n        net = timm.create_model(model_name, pretrained=False, num_classes=0,\n                                img_size=224)\n        state = _torch_load_safe(weight_path)\n        if isinstance(state, dict) and \"state_dict\" in state:\n            state = state[\"state_dict\"]\n        # DINOv2 checkpoints from Meta sometimes have \"teacher.backbone.\" prefix.\n        state = {k.replace(\"teacher.backbone.\", \"\").replace(\"backbone.\", \"\"): v\n                 for k, v in state.items()}\n        # pos_embed shape depends on img_size â€” original 518 â†’ 37Ã—37+1 = 1370\n        # tokens, our 224 â†’ 16Ã—16+1 = 257. Interpolate rather than discard so\n        # DINOv2's pretrained positional info actually reaches the model.\n        if \"pos_embed\" in state:\n            import torch.nn.functional as F\n            pretrained_pe = state[\"pos_embed\"]                # (1, N_old, D)\n            target_shape = net.pos_embed.shape                # (1, N_new, D)\n            if pretrained_pe.shape != target_shape:\n                cls = pretrained_pe[:, :1]                    # (1, 1, D)\n                patch_pe = pretrained_pe[:, 1:]               # (1, N-1, D)\n                old = int(round(patch_pe.shape[1] ** 0.5))\n                new = int(round((target_shape[1] - 1) ** 0.5))\n                patch_pe = (patch_pe.reshape(1, old, old, -1)\n                                    .permute(0, 3, 1, 2))\n                patch_pe = F.interpolate(patch_pe, size=(new, new),\n                                         mode=\"bicubic\", align_corners=False)\n                patch_pe = (patch_pe.permute(0, 2, 3, 1)\n                                    .reshape(1, new * new, -1))\n                state[\"pos_embed\"] = torch.cat([cls, patch_pe], dim=1)\n        missing, unexpected = net.load_state_dict(state, strict=False)\n        if len(unexpected) > 5:\n            print(f\"[cnn] DINOv2: {len(unexpected)} unexpected keys, may indicate wrong variant\")\n    except Exception as e:\n        print(f\"[cnn] DINOv2 load failed ({e}); trying ResNet18 next.\")\n        return False\n    net.eval()\n    device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n    net = net.to(device)\n    _CNN.model = net\n    _CNN.device = device\n    _CNN.feature_size = feat_dim\n    _CNN.feature_names = [f\"dv2{i:04d}\" for i in range(feat_dim)]\n    _CNN.mode = f\"dinov2_{tag}\"\n    # DINOv2 uses ImageNet norm at 224 (or multiple-of-14 resolution).\n    _CNN.transform = transforms.Compose([\n        transforms.ToPILImage(),\n        transforms.Resize((224, 224)),\n        transforms.ToTensor(),\n        transforms.Normalize(mean=[0.485, 0.456, 0.406],\n                             std=[0.229, 0.224, 0.225]),\n    ])\n    print(f\"[cnn] enabled: DINOv2/{tag} @ {device}, weights={weight_path.name}, \"\n          f\"feature_size={feat_dim}\")\n    return True\n\n\ndef _try_setup_resnet18(torch, transforms) -> bool:\n    from torchvision import models\n    weight_paths = list(Path(\"/kaggle/input\").rglob(\"resnet18*.pth\"))\n    if not weight_paths:\n        return False\n    weight_path = weight_paths[0]\n    try:\n        net = models.resnet18(weights=None)\n        state = _torch_load_safe(weight_path)\n        if isinstance(state, dict) and \"state_dict\" in state:\n            state = state[\"state_dict\"]\n        net.load_state_dict(state, strict=False)\n    except Exception as e:\n        print(f\"[cnn] ResNet18 load failed ({e}); using pixel_stats.\")\n        return False\n    net.fc = torch.nn.Identity()\n    net.eval()\n    device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n    net = net.to(device)\n    _CNN.model = net\n    _CNN.device = device\n    _CNN.feature_size = 512\n    _CNN.feature_names = [f\"r18_{i:03d}\" for i in range(512)]\n    _CNN.mode = \"resnet18\"\n    _CNN.transform = transforms.Compose([\n        transforms.ToPILImage(),\n        transforms.Resize((224, 224)),\n        transforms.Grayscale(num_output_channels=3),\n        transforms.ToTensor(),\n        transforms.Normalize(mean=[0.485, 0.456, 0.406],\n                             std=[0.229, 0.224, 0.225]),\n    ])\n    print(f\"[cnn] enabled: ResNet18 @ {device}, weights={weight_path.name}, feature_size=512\")\n    return True\n\n\ndef _setup_cnn() -> None:\n    \"\"\"Prefer DINOv2 â†’ ResNet18 â†’ pixel_stats. Silent no-op on any failure.\"\"\"\n    try:\n        import torch\n        from torchvision import transforms\n    except Exception as e:\n        print(f\"[cnn] torch/torchvision unavailable ({e}); using pixel_stats.\")\n        return\n    if _try_setup_dinov2(torch, transforms):\n        return\n    if _try_setup_resnet18(torch, transforms):\n        return\n    print(\"[cnn] no pretrained weights found under /kaggle/input; using pixel_stats.\")\n\n\ndef _extract_slice_features_batch(imgs_3ch: list[np.ndarray]) -> np.ndarray:\n    \"\"\"\n    imgs_3ch: list of (H, W, 3) float32 in [0,1]. Returns (N, feature_size).\n    Uses the CNN when active; pixel_stats (on the center channel) otherwise.\n    \"\"\"\n    if not imgs_3ch:\n        return np.zeros((0, _CNN.feature_size), dtype=np.float32)\n    if _CNN.mode.startswith((\"dinov2\", \"resnet18\")):\n        import torch\n        tensors = []\n        for im in imgs_3ch:\n            u8 = np.clip(im * 255.0, 0, 255).astype(np.uint8)  # (H, W, 3)\n            tensors.append(_CNN.transform(u8))\n        batch = torch.stack(tensors, dim=0).to(_CNN.device)\n        with torch.no_grad():\n            feats = _CNN.model(batch)\n        return feats.detach().cpu().numpy().astype(np.float32)\n    # Pixel-stats fallback uses only the center (green) channel.\n    return np.stack([_slice_features(im[:, :, 1]) for im in imgs_3ch],\n                    axis=0).astype(np.float32)\n\n\ndef _series_ordered_paths(series_dir: Path) -> list[Path]:\n    \"\"\"Sort DICOM files by physical through-plane position (not filename).\n\n    Filenames are SOP Instance UIDs â€” spearman(filename_rank, physical\n    position) â‰ˆ 0.009 measured over this corpus. Sorting by them picks\n    random slices for 2.5D neighbours and 'start/middle/end' sampling.\n\n    Correct order: project ImagePositionPatient onto the slice normal\n    (cross of the two ImageOrientationPatient vectors). Falls back to\n    InstanceNumber, then natural filename sort.\n    \"\"\"\n    files = list(series_dir.glob(\"*.dcm\"))\n    if not files:\n        return []\n    keyed = []\n    fallback_used = 0\n    for f in files:\n        k = None\n        try:\n            ds = pydicom.dcmread(str(f), force=True, stop_before_pixels=True,\n                                 specific_tags=[\"ImagePositionPatient\",\n                                                \"ImageOrientationPatient\",\n                                                \"InstanceNumber\"])\n            iop = getattr(ds, \"ImageOrientationPatient\", None)\n            ipp = getattr(ds, \"ImagePositionPatient\", None)\n            if iop is not None and ipp is not None and len(iop) >= 6 and len(ipp) >= 3:\n                iop_arr = np.asarray(iop[:6], dtype=np.float64)\n                ipp_arr = np.asarray(ipp[:3], dtype=np.float64)\n                n = np.cross(iop_arr[:3], iop_arr[3:])\n                k = float(np.dot(ipp_arr, n))\n            if k is None or not np.isfinite(k):\n                inst = getattr(ds, \"InstanceNumber\", None)\n                if inst is not None:\n                    k = float(inst)\n                    fallback_used += 1\n        except Exception:\n            pass\n        keyed.append((k, f))\n    # If most slices lack geometry, fall back to natural filename sort\n    if sum(1 for k, _ in keyed if k is None) > len(keyed) // 2:\n        return sorted(files, key=lambda p: p.name)\n    # Missing entries slot to +inf so they end up at the tail rather than\n    # being dropped.\n    return [f for _, f in sorted(keyed, key=lambda t: (t[0] if t[0] is not None else float(\"inf\"), t[1].name))]\n\n\ndef _sample_slice_indices(n: int, k: int) -> list[int]:\n    \"\"\"Sample k slice positions spread over the CENTRAL BAND of a stack.\n\n    Outermost knee slices are mostly soft tissue outside the joint; findings\n    live in the middle. Use 15%..85% central band, matching the 0.899 recipe's\n    default SLICE_BAND.\n    \"\"\"\n    if n <= 0:\n        return []\n    if n <= k:\n        return list(range(n))\n    lo = int(0.15 * (n - 1))\n    hi = int(0.85 * (n - 1))\n    if hi <= lo:\n        return [n // 2] * k\n    return list(dict.fromkeys(int(round(x)) for x in np.linspace(lo, hi, k)))\n\n\n# A small allowlist of safe DICOM header fields. We deliberately do NOT feed\n# Manufacturer/model into the model â€” those are the documented shortcut for\n# site memorisation â€” but we still capture them separately for grouped CV.\n_SCANNER_FIELDS = (\"Manufacturer\", \"ManufacturerModelName\", \"SoftwareVersions\")\n\n\ndef _series_dicom_meta(ds) -> dict:\n    \"\"\"Return a dict of {feature_name: value} from one DICOM header.\"\"\"\n    def _num(tag, default=np.nan):\n        v = getattr(ds, tag, None)\n        try:\n            return float(v)\n        except Exception:\n            return default\n\n    return {\n        \"meta_repetition_time\": _num(\"RepetitionTime\"),\n        \"meta_echo_time\": _num(\"EchoTime\"),\n        \"meta_slice_thickness\": _num(\"SliceThickness\"),\n        \"meta_pixel_spacing_0\": _num_from_seq(ds, \"PixelSpacing\", 0),\n        \"meta_pixel_spacing_1\": _num_from_seq(ds, \"PixelSpacing\", 1),\n        \"meta_magnetic_field\": _num(\"MagneticFieldStrength\"),\n        \"meta_imaging_frequency\": _num(\"ImagingFrequency\"),\n    }\n\n\ndef _num_from_seq(ds, tag: str, idx: int) -> float:\n    v = getattr(ds, tag, None)\n    try:\n        return float(v[idx])\n    except Exception:\n        return float(\"nan\")\n\n\ndef _scanner_key(ds) -> str:\n    parts = []\n    for f in _SCANNER_FIELDS:\n        v = getattr(ds, f, None)\n        parts.append(str(v) if v is not None else \"?\")\n    return \"|\".join(parts)\n\n\n# %% [per-study feature extraction] -----------------------------------------\ndef _empty_slice_vec() -> np.ndarray:\n    return np.full(_CNN.feature_size, np.nan, dtype=np.float32)\n\n\ndef _aggregate_series_arr(arr: np.ndarray) -> np.ndarray:\n    \"\"\"Mean + max across slices in one series. Input (N, F). Output (2F,).\n    Max captures the peak per-feature response â€” a nonparametric proxy for\n    MIL attention pooling, which is what top solutions use for the\n    \"abnormality shows up on just one slice\" case.\"\"\"\n    if arr.size == 0:\n        empty = _empty_slice_vec()\n        return np.concatenate([empty, empty])\n    return np.concatenate([np.nanmean(arr, axis=0), np.nanmax(arr, axis=0)])\n\n\ndef _per_series_len() -> int:\n    return 2 * _CNN.feature_size\n\n\ndef _features_for_study(study_dir: Path,\n                        series_rows: pd.DataFrame) -> tuple[dict, str]:\n    \"\"\"\n    Extract a feature dict for one study.\n\n    Returns (features, scanner_key). scanner_key is used only for grouped CV,\n    never as a model input.\n    \"\"\"\n    # Aggregate per-plane so different anatomical views become distinct blocks\n    # in the feature vector.\n    plane_vecs: dict[str, list[np.ndarray]] = {p: [] for p in PLANES}\n    meta_accum: list[dict] = []\n    scanner_key = \"unknown\"\n\n    # Cap the number of series considered per study (long tail exists).\n    considered = series_rows.head(MAX_SERIES_PER_STUDY)\n    for _, row in considered.iterrows():\n        series_dir = study_dir / row[\"SeriesInstanceUID\"]\n        if not series_dir.is_dir():\n            continue\n        # Geometry-sorted paths â€” filenames are anatomically random (see\n        # _series_ordered_paths docstring).\n        ordered = _series_ordered_paths(series_dir)\n        if not ordered:\n            continue\n        # Sample from the central band (findings live there; edges are soft\n        # tissue). For 2.5D we also need each sampled slice's Â±1 neighbours.\n        sampled = _sample_slice_indices(len(ordered), SLICES_PER_SERIES)\n        needed = set(sampled)\n        if USE_25D_INPUT:\n            for i in sampled:\n                needed.add(max(0, i - 1))\n                needed.add(min(len(ordered) - 1, i + 1))\n        # Read raw pixels once per needed slice.\n        raw_by_idx: dict[int, np.ndarray] = {}\n        first_ds = None\n        for i in sorted(needed):\n            ds = _safe_read_dcm(ordered[i])\n            if ds is None:\n                continue\n            if first_ds is None:\n                first_ds = ds\n            img = _pixels_or_none(ds)\n            if img is None:\n                continue\n            raw_by_idx[i] = img\n        if not raw_by_idx:\n            continue\n        if first_ds is not None and scanner_key == \"unknown\":\n            scanner_key = _scanner_key(first_ds)\n            meta_accum.append(_series_dicom_meta(first_ds))\n        # ONE intensity window per series (over all decoded slices), so a\n        # slice through effusion stays visibly brighter than a slice through\n        # bone â€” crucial for Effusion / Synovitis / Baker's contrast.\n        lo, hi = _series_window(list(raw_by_idx.values()))\n        windowed = {i: _apply_window(a, lo, hi) for i, a in raw_by_idx.items()}\n        windowed = _series_hist_equalize(windowed)  # scanner-invariant CDF mapping\n        # Build 2.5D triplets (prev / current / next) or replicated grayscale.\n        slice_inputs: list[np.ndarray] = []\n        for i in sampled:\n            if i not in windowed:\n                continue\n            curr = windowed[i]\n            if USE_25D_INPUT:\n                prev = windowed.get(max(0, i - 1), curr)\n                nxt = windowed.get(min(len(ordered) - 1, i + 1), curr)\n                if prev.shape == curr.shape and nxt.shape == curr.shape:\n                    slice_inputs.append(np.stack([prev, curr, nxt], axis=-1))\n                    continue\n            slice_inputs.append(np.stack([curr, curr, curr], axis=-1))\n        # One batched call â€” GPU inference is only cheap when batched.\n        slice_arr = _extract_slice_features_batch(slice_inputs)\n        series_vec = _aggregate_series_arr(slice_arr)\n        # Fluid_Sensitive / Fat_Suppression / Plane give us obvious buckets.\n        plane = row.get(\"Anatomical_Plane\", \"Sagittal\")\n        if plane not in plane_vecs:\n            plane = \"Sagittal\"\n        plane_vecs[plane].append(series_vec)\n\n    feats: dict[str, float] = {}\n    for plane in PLANES:\n        if plane_vecs[plane]:\n            block = np.nanmean(np.stack(plane_vecs[plane], axis=0), axis=0)\n            block_has = 1.0\n        else:\n            block = np.full(_per_series_len(), np.nan, dtype=np.float32)\n            block_has = 0.0\n        for i, v in enumerate(block):\n            feats[f\"{plane}_{i:02d}\"] = float(v)\n        # `{plane}_present` was also protocol-correlated (some scanners always\n        # image all 3 planes, others don't) â€” gate it under the same flag.\n        if USE_PROTOCOL_COUNT_FEATURES:\n            feats[f\"{plane}_present\"] = block_has\n\n    # Protocol-count features â€” leaky per v6 grouped-vs-random CV analysis\n    # (see USE_PROTOCOL_COUNT_FEATURES comment in the config block). Off by\n    # default in v7.\n    if USE_PROTOCOL_COUNT_FEATURES:\n        feats[\"n_series\"] = float(len(considered))\n        for p in PLANES:\n            feats[f\"n_series_{p}\"] = float((considered[\"Anatomical_Plane\"] == p).sum())\n        feats[\"n_fluid_sensitive\"] = float(considered.get(\"Fluid_Sensitive\", pd.Series(dtype=int)).fillna(0).sum())\n        feats[\"n_fat_suppression\"] = float(considered.get(\"Fat_Suppression\", pd.Series(dtype=int)).fillna(0).sum())\n\n    # DICOM header timing/geometry features â€” DISABLED by default because\n    # v5 CV showed they leak scanner identity (see USE_SCANNER_METADATA_FEATURES\n    # comment in the config block). We still collect them so scanner_key is\n    # populated for the grouped-CV split; we just don't feed them to the model.\n    if USE_SCANNER_METADATA_FEATURES:\n        if meta_accum:\n            meta_df = pd.DataFrame(meta_accum)\n            for c in meta_df.columns:\n                feats[c] = float(meta_df[c].mean(skipna=True))\n        else:\n            for c in (\"meta_repetition_time\", \"meta_echo_time\", \"meta_slice_thickness\",\n                      \"meta_pixel_spacing_0\", \"meta_pixel_spacing_1\",\n                      \"meta_magnetic_field\", \"meta_imaging_frequency\"):\n                feats[c] = float(\"nan\")\n\n    return feats, scanner_key\n\n\ndef _build_feature_frame(studies_csv: pd.DataFrame,\n                         series_csv: pd.DataFrame,\n                         root: Path,\n                         subsample: int | None = None) -> tuple[pd.DataFrame, pd.Series]:\n    if subsample:\n        studies_csv = studies_csv.sample(\n            n=min(subsample, len(studies_csv)),\n            random_state=RANDOM_STATE,\n        ).reset_index(drop=True)\n    series_by_study = series_csv.groupby(\"StudyInstanceUID\")\n\n    rows, scanners = [], []\n    t0 = time.time()\n    for i, sid in enumerate(studies_csv[\"StudyInstanceUID\"].tolist()):\n        study_dir = root / sid\n        try:\n            srows = series_by_study.get_group(sid)\n        except KeyError:\n            srows = pd.DataFrame(columns=series_csv.columns)\n        feats, sk = _features_for_study(study_dir, srows)\n        feats[\"StudyInstanceUID\"] = sid\n        rows.append(feats)\n        scanners.append(sk)\n        if (i + 1) % 25 == 0 or i + 1 == len(studies_csv):\n            elapsed = time.time() - t0\n            print(f\"  [{i+1}/{len(studies_csv)}] studies processed \"\n                  f\"({elapsed:.1f}s, {(i+1)/max(elapsed, 1e-6):.2f} studies/s)\")\n    df = pd.DataFrame(rows).set_index(\"StudyInstanceUID\")\n    return df, pd.Series(scanners, index=df.index, name=\"scanner_key\")\n\n\n# %% [training / validation] ------------------------------------------------\nfrom sklearn.ensemble import HistGradientBoostingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import StratifiedKFold, GroupKFold\n\n\ndef _fit_predict_per_label(X_train: pd.DataFrame,\n                           Y_train: pd.DataFrame,\n                           X_test: pd.DataFrame,\n                           sample_weight: np.ndarray | None = None) -> np.ndarray:\n    \"\"\"Train per-label models; return P(test, 12).\n\n    v7: HGB + LogisticRegression 60/40 ensemble when USE_LR_ENSEMBLE=True.\n    LR generalises better under scanner distribution shift because trees\n    over-memorise feature-value combinations that map to sites.\n\n    sample_weight up-weights expert-labeled rows over weak report-derived rows.\n    \"\"\"\n    preds = np.full((len(X_test), len(LABEL_COLS)), 0.5, dtype=np.float32)\n\n    # LR needs scaled + NaN-free inputs. Fit scaler once on the pooled train\n    # matrix so per-label training is consistent.\n    if USE_LR_ENSEMBLE:\n        Xtr_lr = np.nan_to_num(X_train.values, nan=0.0).astype(np.float32)\n        Xte_lr = np.nan_to_num(X_test.values, nan=0.0).astype(np.float32)\n        scaler = StandardScaler()\n        scaler.fit(Xtr_lr)\n        Xtr_lr = scaler.transform(Xtr_lr)\n        Xte_lr = scaler.transform(Xte_lr)\n\n    for j, col in enumerate(LABEL_COLS):\n        y = Y_train[col].astype(int).values\n        if y.sum() == 0 or y.sum() == len(y):\n            preds[:, j] = float(y.mean())\n            continue\n        # HGB â€” handles NaN natively; non-linear on raw features.\n        hgb = HistGradientBoostingClassifier(\n            max_iter=300, max_depth=6, learning_rate=0.05,\n            l2_regularization=1.0, random_state=RANDOM_STATE,\n        )\n        hgb.fit(X_train.values, y, sample_weight=sample_weight)\n        p_hgb = hgb.predict_proba(X_test.values)[:, 1]\n        if not USE_LR_ENSEMBLE:\n            preds[:, j] = p_hgb\n            continue\n        # LR â€” L2-regularised, class-weight balanced. Strong linear baseline\n        # that generalises better than HGB when scanner shift dominates.\n        try:\n            lr = LogisticRegression(C=0.1, max_iter=1000,\n                                    class_weight=\"balanced\",\n                                    solver=\"lbfgs\", random_state=RANDOM_STATE)\n            lr.fit(Xtr_lr, y, sample_weight=sample_weight)\n            p_lr = lr.predict_proba(Xte_lr)[:, 1]\n        except Exception:\n            p_lr = p_hgb   # graceful fallback if LR fails to converge etc.\n        preds[:, j] = 0.6 * p_hgb + 0.4 * p_lr\n    return preds\n\n\ndef _macro_auc(y_true: pd.DataFrame, y_pred: np.ndarray) -> tuple[float, dict]:\n    per_label: dict[str, float] = {}\n    for j, col in enumerate(LABEL_COLS):\n        yt = y_true[col].astype(int).values\n        if yt.sum() == 0 or yt.sum() == len(yt):\n            per_label[col] = float(\"nan\")\n            continue\n        per_label[col] = float(roc_auc_score(yt, y_pred[:, j]))\n    macro = float(np.nanmean(list(per_label.values())))\n    return macro, per_label\n\n\ndef _cv_score(X: pd.DataFrame, Y: pd.DataFrame,\n              splitter, groups=None,\n              sample_weight: np.ndarray | None = None) -> tuple[float, list[float]]:\n    fold_aucs: list[float] = []\n    oof = np.zeros((len(X), len(LABEL_COLS)), dtype=np.float32)\n    split_args = (X, Y.sum(axis=1) > 0) if groups is None else (X, Y.sum(axis=1) > 0, groups)\n    for fold, (tr, va) in enumerate(splitter.split(*split_args)):\n        sw = sample_weight[tr] if sample_weight is not None else None\n        preds = _fit_predict_per_label(X.iloc[tr], Y.iloc[tr], X.iloc[va],\n                                       sample_weight=sw)\n        oof[va] = preds\n        m, _ = _macro_auc(Y.iloc[va], preds)\n        fold_aucs.append(m)\n        print(f\"    fold {fold+1}: macro AUC = {m:.4f}\")\n    overall, per_label = _macro_auc(Y, oof)\n    print(f\"    overall OOF macro AUC = {overall:.4f}\")\n    for col, v in per_label.items():\n        print(f\"      {col:>17s}: {v:.4f}\")\n    return overall, fold_aucs\n\n\n# %% [main] -----------------------------------------------------------------\ndef main() -> None:\n    print(\"== RSNA Knee Baseline ==\")\n    print(f\"DATA_DIR = {DATA_DIR}\")\n    print(f\"OUT_DIR  = {OUT_DIR}\")\n    _setup_cnn()  # ResNet18 GPU features when weights are attached; else pixel_stats\n    print(f\"Slice feature mode = {_CNN.mode} (feature_size={_CNN.feature_size})\")\n    _try_setup_llm_parser()  # multilingual sentence-transformer for report labels\n    print(f\"LLM parser mode    = {_LLM['mode']}\")\n\n    required = [\"train.csv\", \"train_series.csv\",\n                \"test.csv\", \"test_series.csv\", \"sample_submission.csv\"]\n    missing = [f for f in required if not (DATA_DIR / f).is_file()]\n    if missing:\n        print(f\"ERROR: could not find {missing} under {DATA_DIR}.\")\n        print(\"Contents of /kaggle/input (top level):\")\n        try:\n            for p in Path(\"/kaggle/input\").iterdir():\n                print(f\"  {p}\")\n        except Exception:\n            pass\n        raise FileNotFoundError(\n            f\"Missing {missing} in {DATA_DIR}. Set RSNA_DATA_DIR env var \"\n            \"to the directory that contains train.csv.\"\n        )\n\n    train_csv = pd.read_csv(DATA_DIR / \"train.csv\")\n    train_series = pd.read_csv(DATA_DIR / \"train_series.csv\")\n    test_csv = pd.read_csv(DATA_DIR / \"test.csv\")\n    test_series = pd.read_csv(DATA_DIR / \"test_series.csv\")\n    sample_sub = pd.read_csv(DATA_DIR / \"sample_submission.csv\")\n\n    # --- 1. Brief EDA --------------------------------------------------\n    print(f\"\\ntrain.csv: {train_csv.shape}, cols={list(train_csv.columns)[:6]}...\")\n    print(\"Per-label positive rate:\")\n    for c in LABEL_COLS:\n        if c in train_csv.columns:\n            print(f\"  {c:>17s}: {train_csv[c].mean():.3f} \"\n                  f\"({int(train_csv[c].sum())} positives)\")\n    print(f\"\\ntrain_series.csv: {train_series.shape}\")\n    if \"Anatomical_Plane\" in train_series.columns:\n        print(\"  planes:\", train_series[\"Anatomical_Plane\"].value_counts().to_dict())\n    series_per_study = train_series.groupby(\"StudyInstanceUID\").size()\n    print(f\"  series per study: median={series_per_study.median():.0f}, \"\n          f\"p95={series_per_study.quantile(0.95):.0f}, \"\n          f\"max={series_per_study.max():.0f}\")\n    if \"PatientSex\" in train_csv.columns:\n        miss = train_csv[\"PatientSex\"].isna().sum() + (train_csv[\"PatientSex\"] == \"\").sum()\n        print(f\"  PatientSex missing/blank: {miss}/{len(train_csv)}\")\n    else:\n        print(\"  PatientSex column absent (schema notes warn this can happen)\")\n\n    # --- 1b. Build labels: expert rows + weak labels from Report ------\n    # This is the highest-leverage step. Without it, ~4349 of 4407 rows\n    # have NaN labels that fillna(0) turns into fabricated negatives,\n    # which caps validation AUC around 0.56 no matter what features we use.\n    all_labels, is_expert, label_conf = _merge_expert_and_report_labels(train_csv)\n    n_expert = int(is_expert.sum())\n    n_weak = int((~is_expert & (all_labels.sum(axis=1) > 0)).sum())\n    n_empty = int((~is_expert & (all_labels.sum(axis=1) == 0)).sum())\n    print(f\"\\nLabel coverage after merging Report-derived weak labels:\")\n    print(f\"  expert-labeled rows      : {n_expert}\")\n    print(f\"  report-derived (any pos) : {n_weak}\")\n    print(f\"  no-signal rows (all zero): {n_empty}  (kept as negatives)\")\n    print(\"  merged per-label positive rate:\")\n    for c in LABEL_COLS:\n        print(f\"    {c:>17s}: {all_labels[c].mean():.3f} \"\n              f\"({int(all_labels[c].sum())} positives)\")\n\n    # --- 2/E2E. End-to-end DINOv2 + SlotHead path ----------------------\n    if USE_END2END:\n        try:\n            from end2end import train_and_predict as _e2e_run  # local file\n        except Exception:\n            # Concatenated-cell fallback (notebook builder inlines the module).\n            _e2e_run = train_and_predict  # type: ignore[name-defined]\n\n        # Studies to use for training â€” everything with any label signal\n        # (i.e. drop the ~634 all-zero rows to avoid drowning the loss in noise).\n        train_ids_all = [sid for sid in all_labels.index\n                         if bool(is_expert.loc[sid]) or all_labels.loc[sid].sum() > 0]\n        print(f\"\\n[e2e] training set: {len(train_ids_all)} studies \"\n              f\"({int(is_expert.sum())} expert + rest report-derived)\")\n\n        # Simple holdout for macro-AUC readout (5-fold too expensive here).\n        rng = np.random.default_rng(RANDOM_STATE)\n        perm = rng.permutation(len(train_ids_all))\n        n_hold = int(E2E_HOLDOUT * len(train_ids_all))\n        hold_ids = [train_ids_all[i] for i in perm[:n_hold]]\n        fit_ids = [train_ids_all[i] for i in perm[n_hold:]]\n        test_ids = list(test_csv[\"StudyInstanceUID\"])\n\n        # First: train on `fit_ids`, predict holdout, report macro AUC.\n        print(f\"[e2e] fit on {len(fit_ids)}, holdout {len(hold_ids)}\")\n        hold_preds = _e2e_run(\n            fit_ids, hold_ids,\n            series_df=train_series, root_train=DATA_DIR / \"train_series\",\n            root_test=DATA_DIR / \"train_series\",     # holdout is drawn from train\n            labels_df=all_labels, conf_df=label_conf, is_expert=is_expert,\n            pixel_loader=_pixels_or_none,\n            ordered_paths_fn=_series_ordered_paths,\n            sampler_fn=_sample_slice_indices,\n            series_window_fn=_series_window,\n            apply_window_fn=_apply_window,\n        )\n        hold_y = all_labels.reindex(hold_ids).values\n        per_label = []\n        for j, col in enumerate(LABEL_COLS):\n            y = hold_y[:, j].astype(int)\n            if y.sum() == 0 or y.sum() == len(y):\n                continue\n            per_label.append(roc_auc_score(y, hold_preds[:, j]))\n        holdout_auc = float(np.mean(per_label)) if per_label else float(\"nan\")\n        print(f\"[e2e] holdout macro AUC = {holdout_auc:.4f} \"\n              f\"(over {len(per_label)}/12 labels with both classes present)\")\n\n        # Then: train on ALL data (fit + hold) and predict real test.\n        print(f\"[e2e] final fit on all {len(train_ids_all)} training studies\")\n        test_preds_arr = _e2e_run(\n            train_ids_all, test_ids,\n            series_df=pd.concat([train_series, test_series], ignore_index=True),\n            root_train=DATA_DIR / \"train_series\",\n            root_test=DATA_DIR / \"test_series\",\n            labels_df=all_labels, conf_df=label_conf, is_expert=is_expert,\n            pixel_loader=_pixels_or_none,\n            ordered_paths_fn=_series_ordered_paths,\n            sampler_fn=_sample_slice_indices,\n            series_window_fn=_series_window,\n            apply_window_fn=_apply_window,\n        )\n        pred_df = pd.DataFrame(test_preds_arr, columns=LABEL_COLS, index=test_ids)\n        random_auc = holdout_auc\n        grouped_auc = float(\"nan\")   # end-to-end skips scanner-grouped CV for now\n        _write_submission(pred_df, sample_sub, all_labels)\n        _print_e2e_summary(random_auc)\n        return\n\n    # --- 2. Feature extraction ----------------------------------------\n    print(\"\\nExtracting TRAIN features...\")\n    X_train, scanner_train = _build_feature_frame(\n        train_csv, train_series,\n        root=DATA_DIR / \"train_series\",\n        subsample=TRAIN_SUBSAMPLE,\n    )\n    Y_train = all_labels.reindex(X_train.index).fillna(0).astype(int)\n    # Sample weight combines two signals: a 3Ã— base multiplier for expert rows\n    # (ground truth vs weak labels) AND the parser's per-row confidence\n    # (weak rows where the parser was highly confident count more than rows\n    # where it was hedging). Sample weight is scalar per row, so we take the\n    # mean confidence across the 12 labels for that row.\n    expert_flag = is_expert.reindex(X_train.index).fillna(False).values\n    if USE_ADVANCED_PARSER:\n        row_conf = label_conf.reindex(X_train.index).fillna(0.05).mean(axis=1).values\n        base = np.where(expert_flag, 3.0, 1.0)\n        sample_weight = (base * row_conf).astype(np.float32)\n    else:\n        # v10 path: flat 3Ã— expert, 1Ã— weak â€” no per-row confidence.\n        sample_weight = np.where(expert_flag, 3.0, 1.0).astype(np.float32)\n\n    print(\"\\nExtracting TEST features...\")\n    X_test, _ = _build_feature_frame(\n        test_csv, test_series,\n        root=DATA_DIR / \"test_series\",\n        subsample=None,\n    )\n\n    # Align columns; HistGradientBoosting handles NaN natively.\n    all_cols = sorted(set(X_train.columns) | set(X_test.columns))\n    X_train = X_train.reindex(columns=all_cols)\n    X_test = X_test.reindex(columns=all_cols)\n\n    # --- 3. Validation: random-stratified + scanner-grouped -----------\n    print(\"\\nCV #1: random StratifiedKFold on (has-any-label)...\")\n    skf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=RANDOM_STATE)\n    random_auc, _ = _cv_score(X_train, Y_train, skf, groups=None,\n                              sample_weight=sample_weight)\n\n    print(\"\\nCV #2: GroupKFold by scanner (Manufacturer|Model|Software)...\")\n    n_groups = scanner_train.nunique()\n    if n_groups >= 2:\n        n_splits = int(min(N_FOLDS, n_groups))\n        gkf = GroupKFold(n_splits=n_splits)\n        grouped_auc, _ = _cv_score(X_train, Y_train, gkf,\n                                   groups=scanner_train.values,\n                                   sample_weight=sample_weight)\n    else:\n        grouped_auc = float(\"nan\")\n        print(\"  Only one scanner group present â€” grouped CV skipped.\")\n\n    if not math.isnan(grouped_auc) and (random_auc - grouped_auc) > 0.05:\n        print(\"\\n  WARNING: random-fold AUC >> grouped-fold AUC â€” the model is \"\n              \"leaning on scanner/site identity. Prune metadata features or add\"\n              \" site-invariant training before trusting the leaderboard.\")\n\n    # --- 4. Final fit on all training data, predict on test -----------\n    print(\"\\nFinal fit on all training studies, predicting test...\")\n    test_preds = _fit_predict_per_label(X_train, Y_train, X_test,\n                                        sample_weight=sample_weight)\n    pred_df = pd.DataFrame(test_preds, columns=LABEL_COLS, index=X_test.index)\n\n    # --- 5. Build submission matching sample_submission.csv exactly ---\n    sub = sample_sub.copy()\n    id_col = \"StudyInstanceUID\"\n    pred_df = pred_df.reindex(sub[id_col].values)\n    for c in LABEL_COLS:\n        vals = pred_df[c].values\n        # Any study we couldn't score falls back to the label prior â€” better\n        # than 0.5 and keeps AUC well-defined.\n        prior = float(Y_train[c].mean()) if c in Y_train else 0.5\n        vals = np.where(np.isnan(vals), prior, vals)\n        sub[c] = np.clip(vals, 0.0, 1.0).astype(np.float32)\n\n    sub_path = OUT_DIR / \"submission.csv\"\n    sub.to_csv(sub_path, index=False)\n    print(f\"\\nWrote {sub_path} â€” {sub.shape[0]} rows, {sub.shape[1]} cols\")\n\n    # --- 6. Plain-language summary ------------------------------------\n    print(\"\\n\" + \"=\" * 66)\n    print(\"SUMMARY\")\n    print(\"=\" * 66)\n    print(f\"Random 5-fold macro ROC AUC : {random_auc:.4f}\")\n    print(f\"Scanner-grouped macro AUC   : {grouped_auc:.4f}\")\n    print(\"\\nModel: per-study features = per-plane pixel statistics \"\n          \"(mean/std/percentiles/histogram/edge density) + safe DICOM-header \"\n          \"numbers â†’ one HistGradientBoostingClassifier per label. Labels \"\n          \"are expert-annotated where available, else derived from the \"\n          \"free-text Report via multilingual keyword + sentence-level \"\n          \"negation; expert rows get 3x sample-weight in training.\")\n    print(\"\\nNext steps to lift the score further, in rough order of value:\")\n    print(\"  1. Replace pixel stats with a pretrained CNN feature extractor\"\n          \" (ResNet18/EfficientNet from an attached torchvision-weights\"\n          \" Kaggle dataset) â€” expect +0.05 to +0.10 macro AUC.\")\n    print(\"  2. Improve report parsing: LLM-based label extraction (offline\"\n          \" model attached as a dataset) is much stronger than regex on\"\n          \" noisy multilingual text.\")\n    print(\"  3. Train separate per-plane sub-models (Sagittal / Coronal /\"\n          \" Axial), blend with per-label optimal weights, and add TTA\"\n          \" (flip + neighbouring-slice averaging).\")\n\n\ndef _write_submission(pred_df, sample_sub, all_labels):\n    \"\"\"Reindex predictions to sample_submission.csv order, clip [0,1], write.\"\"\"\n    sub = sample_sub.copy()\n    id_col = \"StudyInstanceUID\"\n    pred_df = pred_df.reindex(sub[id_col].values)\n    for c in LABEL_COLS:\n        vals = pred_df[c].values\n        prior = float(all_labels[c].mean()) if c in all_labels else 0.5\n        vals = np.where(np.isnan(vals), prior, vals)\n        sub[c] = np.clip(vals, 0.0, 1.0).astype(np.float32)\n    sub_path = OUT_DIR / \"submission.csv\"\n    sub.to_csv(sub_path, index=False)\n    print(f\"\\nWrote {sub_path} â€” {sub.shape[0]} rows, {sub.shape[1]} cols\")\n\n\ndef _print_e2e_summary(holdout_auc):\n    print(\"\\n\" + \"=\" * 66)\n    print(\"SUMMARY (end-to-end DINOv2 + SlotHead)\")\n    print(\"=\" * 66)\n    print(f\"Holdout macro ROC AUC : {holdout_auc:.4f}\")\n    print(\"\\nModel: DINOv2 ViT-S/14 with last 2 blocks unfrozen; 3 slots per study \"\n          \"(Sagittal/Coronal/Axial, longest series per plane), 3 slices per slot \"\n          \"(central-band, geometry-ordered, per-series windowed). SlotHead: per-label \"\n          \"attention over slot embeddings with anatomy prior. Labels are expert + \"\n          \"multilingual-parser-derived; weak rows down-weighted by parser confidence.\")\n    print(\"\\nNext levers (from the 0.899 recipe, not yet in v15):\")\n    print(\"  1. 6-slot scheme (plane Ã— fat-suppression Ã— fluid-weighting) instead of 3.\")\n    print(\"  2. Laterality normalisation (flip right knees to left) via IPP+IOP.\")\n    print(\"  3. Unfreeze 6 backbone blocks + 10 epochs instead of 2 + 3 epochs.\")\n    print(\"  4. 336Ã—336 input resolution (currently 224Ã—224); ~2x compute but +AUC.\")\n\n\ndef _write_fallback_submission() -> None:\n    \"\"\"Guarantee submission.csv exists even if the pipeline crashed early.\"\"\"\n    out = OUT_DIR / \"submission.csv\"\n    if out.is_file():\n        return\n    src = DATA_DIR / \"sample_submission.csv\"\n    if src.is_file():\n        pd.read_csv(src).to_csv(out, index=False)\n        print(f\"Fallback submission written from {src} -> {out}\")\n\n\nif __name__ == \"__main__\":\n    try:\n        main()\n    except Exception:\n        import traceback\n        traceback.print_exc()\n        _write_fallback_submission()\n        raise\n"}]}