{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12"},"rsna_optimization":{"base_public_score":0.906,"base_version":"V52","candidate":"uniform_035","hidden_test_safe":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"5219d78f-8a10-4995-a999-0924dcd6a35d","cell_type":"markdown","source":"# RSNA Knee hidden-test-safe inference candidate\n\nGenerated from the verified public-score V52 notebook, keeping its packaged DINOv2\ninference path and adding the public, hash-pinned RadImageNet E10 arm.  Configuration:\n`uniform_035`.  The scored run reads every hidden test DICOM; no visible-test prediction\nor UID is embedded in this notebook.\n","metadata":{}},{"id":"b5a632d3","cell_type":"code","source":"from __future__ import annotations\n\nimport os\n\nfor _v in (\"OMP_NUM_THREADS\", \"OPENBLAS_NUM_THREADS\", \"MKL_NUM_THREADS\"):\n    os.environ.setdefault(_v, \"4\")\n\nimport gc\nimport hashlib\nimport json\nimport re\nimport time\nimport traceback\nimport threading\nfrom concurrent.futures import ThreadPoolExecutor\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\n# Use every accelerator that can execute the current PyTorch kernels.  Kaggle can\n# attach a legacy GPU that torch.cuda.is_available() reports as usable even though\n# the image has no compatible convolution kernel.  A real model-shaped probe makes\n# that allocation fail closed to CPU before any ensemble member is consumed.\ndef _cuda_execution_probe(index):\n    dev = torch.device(f\"cuda:{index}\")\n    try:\n        major, minor = torch.cuda.get_device_capability(index)\n        probe = nn.Conv2d(3, 4, kernel_size=3, padding=1).eval().to(dev)\n        with torch.inference_mode():\n            out = probe(torch.zeros((1, 3, 16, 16), device=dev))\n            if tuple(out.shape) != (1, 4, 16, 16):\n                raise RuntimeError(f\"unexpected CUDA probe shape {tuple(out.shape)}\")\n        torch.cuda.synchronize(index)\n        print(f\"cuda:{index} probe PASS (compute {major}.{minor})\")\n        del probe, out\n        torch.cuda.empty_cache()\n        return True\n    except Exception as exc:\n        print(f\"cuda:{index} probe FAIL ({type(exc).__name__}: {exc}); using CPU fallback\")\n        try:\n            torch.cuda.empty_cache()\n        except Exception:\n            pass\n        return False\n\nDEVS = []\nif torch.cuda.is_available():\n    DEVS = [torch.device(f\"cuda:{i}\") for i in range(torch.cuda.device_count())\n            if _cuda_execution_probe(i)]\nif not DEVS:\n    DEVS = [torch.device(\"cpu\")]\nprint(f\"devices: {[str(d) for d in DEVS]}\")\n\n\n# The label extractor is defined in the cells above when this runs as a notebook. As a\n# plain script it is imported from the package source, so the two paths share one\n# definition rather than keeping a copy each.\n\nT0 = time.time()\nSEED = 2026\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\n\nTARGETS = [\"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\", \"Medial OA\",\n           \"Lateral OA\", \"PF OA\", \"Effusion\", \"Synovitis\", \"Baker's\",\n           \"Contusion\", \"Fracture\"]\n\n\n# The centre crop has to be smaller than the smallest field of view in the corpus or it\n# silently does nothing. Measured over every training series, the acquired field of view\n# (Rows x PixelSpacing) has median 160 mm and runs from 70 to 320: a 160 mm crop is\n# larger than the image in 60% of series and is skipped for all of them, which leaves\n# their physical scale unnormalised. 130 mm is below the field of view of 99.6% of\n# series and still contains the joint.\nCROP_MM = 130.0\n\n# Cache resolution. Everything downstream may downsample from this, so it is set by the\n# most demanding configuration rather than by the default one.\nCACHE_IMG = 336\nGROUP = 3                  # slices per encoder input, stacked as the three channels\nN_GROUP_MAX = 1\nCACHE_FRACTION = 0.45      # share of free memory the pixel cache may take\nCACHE_BUDGET_MAX_GB = 24.0 # hard ceiling regardless of what the machine reports\nCACHE_BUDGET_GB = 12.0     # only the fallback, for a machine with no /proc/meminfo\nTEST_SHARE = 0.30          # floor on the test corpus relative to the training one, since\n                           # the visible test split is a stub and the scored one is not\nHDR_THREADS = 16\nPIX_THREADS = 12\nORDER_THREADS = 32         # slice-ordering is latency-bound on the mount, not CPU-bound\n# Ceiling for the ordering pass. It has to be a ceiling because the pass is hundreds of\n# thousands of small reads over a network mount, so its duration is a property of the\n# mount rather than of the work, and varies between runs that do the same reading. It must not be a tight one: giving up leaves\n# those series in file order, which is uncorrelated with anatomy, and that degradation is\n# silent. So the ceiling sits well above what the pass ordinarily needs: its purpose is\n# to stop the pass consuming the whole run on a slow mount, not to trim the ordinary\n# case, and a ceiling tight enough to bind on a normal day would trade a silent\n# degradation for a saving the run does not need.\nORDER_BUDGET_S = 5400\n\n# Resolution is the axis under test. A feature of width d mm survives resampling only if\n# the pixel pitch is at most d/2, and the pitch here is set by the crop above rather than\n# by the acquired field of view: CROP_MM / P. At 224 px that is 0.58 mm, above the 0.5 mm\n# a 1 mm tear needs; at 336 px it is 0.39 mm and clears it. Both configurations read the\n# same cache, so the comparison isolates the resize.\nRUNS = [\n    {\"name\": \"r224\", \"img\": 224},\n    {\"name\": \"r336\", \"img\": 336},\n]\n\nEPOCHS = 10\nBATCH_STUDIES = 8          # a study is a bag of up to N_SLOT slot images\nAUG_ROT_DEG = 8.0          # rigid jitter; see augment() for why neither flip is used\nAUG_SCALE = 0.08\nAUG_SHIFT = 0.05\nAUG_INTENSITY = 0.10\nLAT_MIN_OFFSET_MM = 20.0   # inside this the side is not readable from geometry; see\n                           # side_from_geometry()\nSLICE_BAND = (0.20, 0.80)  # fraction of the ordered stack read_slot samples across\n\n# --- What a slice IS, as opposed to how many of them there are --------------- #\n#\n# A member is a function of the pixels it was fitted on, and img/crop_mm/slices/band do\n# not determine those pixels by themselves. Four further decisions do, none of them\n# visible in any shape:\n#\n#   order          which slice is the next one along the stack\n#   lat            which knees are mirrored, and on what evidence\n#   slot_fallback  whether a T1 slot may be filled from a series that is not T1\n#   decode_fill    what stands in for a slice that would not decode\n#\n# `native` is the reading derived in the sections below. `legacy` is the reading an\n# imported member was fitted under. A member read under the wrong one loads with every\n# shape matching, runs, and writes a plausible submission computed from the wrong image -\n# so the choice travels with the member and is part of the key that decides which members\n# can share a decode. The legacy rules are reproduced rather than corrected: correcting\n# them would hand that member pixels its weights never saw.\nRULES_NATIVE = {\"order\": \"normal\", \"lat\": \"centre\",\n                \"slot_fallback\": False, \"decode_fill\": \"nearest\"}\nRULES_LEGACY = {\"order\": \"dominant_axis\", \"lat\": \"corner_x\",\n                \"slot_fallback\": True, \"decode_fill\": \"zero\"}\nRULES = dict(RULES_NATIVE)\nLEGACY_LAT_OFFSET_MM = 5.0   # the dead zone the legacy laterality rule was fitted with\n\nLR_HEAD = 1e-3\nLR_BACKBONE = 8e-6         # the encoder is adapted, not retrained\nUNFREEZE_LAST = 6          # trainable transformer blocks, from the output end\nWEIGHT_DECAY = 0.02\nEVAL_BATCH = 8\nTIME_BUDGET = 8.0 * 3600\n\n# Six slots: three planes crossed with the acquisition axes. The fat-suppressed\n# fluid-sensitive series exist for nearly every study; the T1 and the non-suppressed\n# fluid-sensitive series are scarcer, which is what the presence mask is for.\nSLOTS_RECOVERED = [\n    (\"SAG_FLUID_FS\", \"Sagittal\", True, True),\n    (\"COR_FLUID_FS\", \"Coronal\", True, True),\n    (\"AX_FLUID_FS\", \"Axial\", True, True),\n    (\"SAG_FLUID_NOFS\", \"Sagittal\", True, False),\n    (\"COR_T1\", \"Coronal\", False, False),\n    (\"SAG_T1\", \"Sagittal\", False, False),\n]\n\n# The alternative: plane x the single axis the delivered flags carry, ignoring the\n# recovered weighting. Kept as\n# a switch so the choice of slot definition can be varied while everything else is held\n# fixed. Under this scheme a `Struct` slot mixes T1 series with non-fat-suppressed PD/T2\n# series, which carry very different tissue contrast.\nSLOTS_PUBLIC = [\n    (\"SAG_FLUID\", \"Sagittal\", None, True),\n    (\"COR_FLUID\", \"Coronal\", None, True),\n    (\"AX_FLUID\", \"Axial\", None, True),\n    (\"SAG_STRUCT\", \"Sagittal\", None, False),\n    (\"COR_STRUCT\", \"Coronal\", None, False),\n    (\"AX_STRUCT\", \"Axial\", None, False),\n]\n\nSLOT_SCHEME = os.environ.get(\"SLOT_SCHEME\", \"recovered\")\nSLOTS = SLOTS_PUBLIC if SLOT_SCHEME == \"public\" else SLOTS_RECOVERED\nN_SLOT = len(SLOTS)\n\n# How many 384-wide parts the per-slot feature is built from. The encoder emits one\n# vector per token; a slot feature is a fixed summary of that grid, and the summary an\n# imported member was fitted with carries a third part.\nPOOL_PARTS = {\"cls_mean\": 2, \"cls_mean_focal\": 3}\n\n# Which slots an imported member's attention is tilted toward, per diagnosis. Indices are\n# into SLOTS. This is a fixed table rather than a learned parameter, so it is part of that\n# member's definition and has to be reproduced exactly for its weights to mean anything.\nSLOT_PRIOR_TABLE = {\n    \"ACL\": (0, 3, 5), \"MCL\": (1, 4),\n    \"Medial Meniscus\": (0, 1, 3, 4), \"Lateral Meniscus\": (0, 1, 3, 4),\n    \"Medial OA\": (1, 4, 5), \"Lateral OA\": (1, 4, 5),\n    \"PF OA\": (0, 2, 5), \"Effusion\": (0, 2), \"Synovitis\": (0, 2),\n    \"Baker's\": (0,), \"Contusion\": (0, 1, 2), \"Fracture\": (0, 1, 2, 4, 5),\n}\nSLOT_PRIOR_STRENGTH = 0.55\n\nFATSAT_OPTS = {\"FS\", \"FATSAT\", \"FAT_SAT\", \"FSAT\"}\n_SEP = re.compile(r\"[_\\-.]\")\n_FATSAT_RX = re.compile(r\"\\bfs\\b|fatsat|fat sat|\\bstir\\b|\\bspair\\b|\\bspir\\b|\\bwe\\b|\"\n                        r\"water excit|\\btirm\\b|\\bsting\\b|\\bfatsup\\b\")\n_T1_RX = re.compile(r\"\\bt1\\b|\\bt1w\\b\")\n_T2_RX = re.compile(r\"\\bt2\\b|\\bt2w\\b\")\n_PD_RX = re.compile(r\"\\bpd\\b|\\bpdw\\b|proton|\\bdp\\b|dens\")\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-10T06:00:09.61721Z","iopub.status.busy":"2026-08-10T06:00:09.616975Z","iopub.status.idle":"2026-08-10T06:00:15.597811Z","shell.execute_reply":"2026-08-10T06:00:15.596941Z"},"papermill":{"duration":6.003035,"end_time":"2026-08-10T06:00:15.599404+00:00","exception":false,"start_time":"2026-08-10T06:00:09.596369+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"2946e2a8","cell_type":"code","source":"def log(msg):\n    print(f\"[{time.time() - T0:7.1f}s] {msg}\", flush=True)\n\n\ndef find_root():\n    for c in [Path(\"/kaggle/input/competitions/rsna-knee-abnormality-detection\"),\n              Path(\"/kaggle/input/rsna-knee-abnormality-detection\"),\n              Path(\"data\"), Path(\".\")]:\n        if (c / \"test.csv\").is_file() and (c / \"test_series\").is_dir():\n            return c\n    # last resort: two-level scan, because the mount is nested one deeper than usual\n    base = Path(\"/kaggle/input\")\n    if base.is_dir():\n        for depth1 in sorted(p for p in base.iterdir() if p.is_dir()):\n            for cand in [depth1] + sorted(p for p in depth1.iterdir() if p.is_dir()):\n                if (cand / \"test.csv\").is_file():\n                    return cand\n    raise FileNotFoundError(\n        f\"competition mount not found (cwd {Path.cwd()}); expected a directory holding \"\n        f\"test.csv and test_series/\")\n\n\ndef find_dinov2(variant=\"small\"):\n    \"\"\"Locate a mounted DINOv2 checkpoint directory by variant name.\"\"\"\n    base = Path(\"/kaggle/input\")\n    if not base.is_dir():\n        return None\n    hits = []\n    for root, dirs, files in os.walk(base):\n        dirs[:] = [d for d in dirs if d not in (\"train_series\", \"test_series\")]\n        if \"config.json\" in files and \"dinov2\" in root.lower():\n            hits.append(Path(root))\n    for h in hits:\n        if variant in str(h).lower():\n            return h\n    return hits[0] if hits else None\n\n\nLABEL_COLS = TARGETS + [t + \"__conf\" for t in TARGETS]\n\n\nclass LabelSourceError(RuntimeError):\n    \"\"\"Raised when the labels did not come from where this run intended.\n\n    Every other failure in this file is better survived than reported: a run that dies\n    after the cache is built has spent the expensive half and scores nothing, so the\n    guard around `main` swallows it and leaves the benchmark file behind. This one is\n    the exception. Training on the weaker labels does not look like a failure - it\n    completes, writes a plausible submission, and differs only in a log line - so it has\n    to stop the run rather than be absorbed by a guard designed for crashes.\n    \"\"\"\n\n\ndef find_label_table():\n    \"\"\"Locate a mounted table of pre-read report labels, if one is attached.\n\n    The lexicon turns a report into labels by matching morphology, and its failure\n    mode is silence: on a phrasing it does not carry it emits no opinion rather than a\n    wrong one. Silence is measurable without any ground truth - for each (report,\n    finding) pair, did anything match? - and that measurement says the misses are\n    concentrated in particular languages rather than spread evenly, on findings a knee\n    report almost always comments on.\n\n    Enumerating morphology for nine languages is the wrong instrument for that. Reading\n    the sentence is the right one, and a language model reads it. Against the annotated\n    studies the difference is large and one-sided, so when such a table is mounted it is\n    preferred; when it is not, the lexicon runs and the pipeline is unchanged. Both paths\n    produce the same columns, so nothing downstream knows which one supplied them.\n    \"\"\"\n    base = Path(\"/kaggle/input\")\n    cands = []\n    if base.is_dir():\n        for root, dirs, files in os.walk(base):\n            dirs[:] = [d for d in dirs if d not in (\"train_series\", \"test_series\")]\n            cands += [Path(root) / f for f in files if f.startswith(\"report_labels\")\n                      and f.endswith(\".csv\")]\n    cands += [p for p in (Path(\"data/derived/report_labels_v2.csv\"),) if p.is_file()]\n    for c in cands:\n        try:\n            head = pd.read_csv(c, nrows=1)\n        except Exception:\n            continue\n        if \"StudyInstanceUID\" in head.columns and all(t in head.columns for t in TARGETS):\n            return c\n    return None\n\n\ndef label_mount_attached():\n    \"\"\"True when an input directory was attached that is meant to carry a label table.\n\n    The fallback below is deliberate and has to stay silent for a run with no table\n    attached, because that is the ordinary case for anyone reading this notebook. It\n    must not stay silent for the other case: a table was attached and could not be used.\n    Those two are indistinguishable from the labels alone - both end with the lexicon -\n    so they are separated here by whether the mount exists at all.\n    \"\"\"\n    base = Path(\"/kaggle/input\")\n    if not base.is_dir():\n        return False\n    return any(\"label\" in p.name.lower() for p in base.iterdir() if p.is_dir())\n\n\ndef read_labels(train_df):\n    \"\"\"Labels for every training study, from a mounted table or from the lexicon.\n\n    Studies the mounted table does not cover fall back to the lexicon rather than being\n    dropped, so a partial table degrades coverage instead of losing rows.\n    \"\"\"\n    n = len(train_df)\n    lab = pd.DataFrame([extract(r) for r in train_df[\"Report\"].fillna(\"\")])\n    lab[\"StudyInstanceUID\"] = train_df[\"StudyInstanceUID\"].values\n    lab = lab.set_index(\"StudyInstanceUID\")\n\n    src = find_label_table()\n    if src is None:\n        if label_mount_attached():\n            raise LabelSourceError(\n                \"LABEL SOURCE: a label dataset is mounted but no usable table was found \"\n                \"in it. Falling back to the lexicon here would train on the weaker \"\n                \"labels and say so only in a log line, so the run stops instead.\")\n        log(f\"LABEL SOURCE: lexicon, {n} studies (no table mounted)\")\n        return lab\n\n    tab = pd.read_csv(src).set_index(\"StudyInstanceUID\")\n    missing = [c for c in LABEL_COLS if c not in tab.columns]\n    if missing:\n        raise LabelSourceError(\n            f\"LABEL SOURCE: {src} is missing {len(missing)} expected columns \"\n            f\"(first: {missing[0]!r}). Refusing to fall back silently.\")\n    hit = lab.index.intersection(tab.index)\n    if not len(hit):\n        raise LabelSourceError(\n            f\"LABEL SOURCE: {src} shares no StudyInstanceUID with train.csv.\")\n    log(f\"LABEL SOURCE: {src.name} covers {len(hit)} of {n} studies, \"\n        f\"lexicon for the remaining {n - len(hit)}\")\n    lab.loc[hit, LABEL_COLS] = tab.loc[hit, LABEL_COLS].values\n    return lab\n\n\nROOT = find_root()\nlog(f\"input root: {ROOT}\")\n\n\nIMG = CACHE_IMG            # kept as the name the pixel reader and cache use\n\n\ndef available_gb():\n    \"\"\"Memory this machine will actually lend, read rather than assumed.\n\n    A hardcoded ceiling is a guess about a machine the author is not sitting at, and a\n    guess that is too low costs coverage silently while a guess that is too high ends the\n    run. The machine will say, so it is asked.\n    \"\"\"\n    try:\n        with open(\"/proc/meminfo\") as fh:\n            info = {k.strip(): v for k, v in\n                    (l.split(\":\", 1) for l in fh if \":\" in l)}\n        return int(info[\"MemAvailable\"].split()[0]) / 1024 ** 2\n    except Exception:\n        return CACHE_BUDGET_GB / CACHE_FRACTION      # fall back to the old constant\n\n\ndef plan_cache(n_study, n_test=0):\n    \"\"\"Choose how many slices per slot the memory the machine has will allow.\n\n    The cache is n_study x n_slot x slices x IMG^2 bytes. Coverage is the cheap axis -\n    linear - and resolution the expensive one, so when the budget binds it is the slice\n    count that gives way rather than the pixel grid. Deciding once, from the training\n    corpus size, keeps train and test caches on the same group layout.\n\n    Only a fraction of what is free is taken. The rest is not slack: the encoder, its\n    activations, the pinned batches and the frames all come out of the same pool, and the\n    cache is the one allocation big enough that overshooting it kills the run outright.\n    \"\"\"\n    avail = available_gb()\n    budget = min(avail * CACHE_FRACTION, CACHE_BUDGET_MAX_GB)\n    # Both caches are held at once, and the test half is what the visible run cannot\n    # show: here it is a handful of studies, and at scoring it is the whole hidden set.\n    # Sizing against the training corpus alone therefore passes every run that can be\n    # watched and overruns the one that counts.\n    n_total = n_study + max(n_test, int(TEST_SHARE * n_study))\n    per_slice = n_total * N_SLOT * IMG * IMG\n    afford = int(budget * 1024 ** 3 // max(per_slice, 1))\n    groups = max(1, min(N_GROUP_MAX, afford // GROUP))\n    log(f\"memory: {avail:.1f} GB available, {budget:.1f} GB to the cache; \"\n        f\"sizing for {n_study} train + {n_total - n_study} test studies \"\n        f\"-> {groups} group(s) of {GROUP} = {groups * GROUP} slices per slot\"\n        + (f\" (wanted {N_GROUP_MAX})\" if groups < N_GROUP_MAX else \"\"))\n    return groups\n\n\nN_GROUP = plan_cache(len(pd.read_csv(ROOT / \"train.csv\")),\n                     len(pd.read_csv(ROOT / \"test.csv\")))\nCACHE_SLICES = GROUP * N_GROUP\nlog(f\"cache layout: {N_GROUP} groups x {GROUP} slices = {CACHE_SLICES} per slot\")\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-10T06:00:15.641496Z","iopub.status.busy":"2026-08-10T06:00:15.640853Z","iopub.status.idle":"2026-08-10T06:00:15.771336Z","shell.execute_reply":"2026-08-10T06:00:15.770343Z"},"papermill":{"duration":0.152593,"end_time":"2026-08-10T06:00:15.773287+00:00","exception":false,"start_time":"2026-08-10T06:00:15.620694+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"52287fbf","cell_type":"code","source":"HDR_TAGS = [\"SeriesDescription\", \"SequenceName\", \"ScanOptions\", \"ScanningSequence\",\n            \"RepetitionTime\", \"EchoTime\", \"Laterality\", \"PixelSpacing\", \"Rows\",\n            \"Columns\", \"RescaleSlope\", \"RescaleIntercept\",\n            # Position and orientation are read from the same header probe() already\n            # opens, so they cost nothing, and they are what recovers the side when the\n            # Laterality tag is absent - which it is for half the studies here.\n            \"ImagePositionPatient\", \"ImageOrientationPatient\"]\n\n\ndef _hdr_vec(s, n):\n    \"\"\"Parse a DICOM multi-value string as stored by probe(): floats joined by `|`.\"\"\"\n    if not isinstance(s, str):\n        return None\n    try:\n        v = [float(x) for x in s.split(\"|\")]\n    except ValueError:\n        return None\n    return np.array(v) if len(v) >= n else None\n\n\ndef side_from_geometry(h):\n    \"\"\"Study -> 'L' / 'R' / None, from where the image sits in the patient.\n\n    `Laterality` (0020,0060) is Type 2C and may legitimately be absent; in this corpus it\n    is missing on exactly half the studies, and the vendors it is missing from are whole\n    vendors rather than scattered series. A study with no tag is not a left knee, but the\n    normalisation upstream treats it as one, so half the corpus was never normalised and\n    the five side-defined targets - the two menisci, the two tibiofemoral compartments\n    and the medial collateral ligament - saw that axis reversed on a large minority of it.\n\n    The patient coordinate system fixes this without the tag: +x is the patient's left, so\n    the centre of a right knee sits at negative x. The centre is used rather than\n    `ImagePositionPatient` itself because that is the corner of the image, which is offset\n    by half a field of view - enough to change the sign on a knee near the midline.\n\n    The median over a study's series is what is thresholded, not a single series: probe()\n    reads one arbitrary slice per series, which on a sagittal stack can sit anywhere\n    across the joint. Studies whose centre falls near the midline are left unresolved\n    rather than guessed - measured against the tagged half, the rule is right 97% of the\n    time overall and no better than chance inside 20 mm.\n    \"\"\"\n    cx = {}\n    for r in h.itertuples(index=False):\n        ipp = _hdr_vec(getattr(r, \"ImagePositionPatient\", None), 3)\n        iop = _hdr_vec(getattr(r, \"ImageOrientationPatient\", None), 6)\n        ps = _hdr_vec(getattr(r, \"PixelSpacing\", None), 2)\n        rows, cols = getattr(r, \"Rows\", None), getattr(r, \"Columns\", None)\n        if ipp is None or iop is None or ps is None or not rows or not cols:\n            continue\n        try:\n            c = ipp[:3] + iop[:3] * ps[1] * float(cols) / 2 + iop[3:6] * ps[0] * float(rows) / 2\n        except (TypeError, ValueError):\n            continue\n        cx.setdefault(r.StudyInstanceUID, []).append(float(c[0]))\n    out = {}\n    for st, xs in cx.items():\n        m = float(np.median(xs))\n        out[st] = None if abs(m) < LAT_MIN_OFFSET_MM else (\"R\" if m < 0 else \"L\")\n    return out\n\n\ndef side_from_corner_x(h):\n    \"\"\"The laterality an imported member was fitted under.\n\n    It thresholds the median raw `ImagePositionPatient` x over a study's series. That is\n    the x of the image *corner*, not of its centre, so it differs from the rule above by\n    up to half a field of view - which is enough to reverse the sign on a knee scanned\n    near the midline. The dead zone is 5 mm rather than 20 mm, so it also commits on\n    studies the rule above leaves unresolved.\n\n    Neither difference changes a shape. Each one decides whether a study is mirrored, and\n    a study mirrored one way at training and the other at inference presents the five\n    side-defined targets with their axis reversed.\n    \"\"\"\n    out = {}\n    for st, g in h.groupby(\"StudyInstanceUID\"):\n        xs = []\n        for r in g.itertuples(index=False):\n            ipp = _hdr_vec(getattr(r, \"ImagePositionPatient\", None), 3)\n            if ipp is not None and np.isfinite(ipp).all():\n                xs.append(float(ipp[0]))\n        if not xs:\n            out[st] = None\n            continue\n        x = float(np.median(xs))\n        # DICOM patient coordinates are LPS: +x is the patient's left.\n        out[st] = None if abs(x) < LEGACY_LAT_OFFSET_MM else (\"R\" if x < 0 else \"L\")\n    return out\n\n\ndef lat_of(h, tag=\"\"):\n    \"\"\"Study -> 'L' / 'R' / None: the tag where it exists, geometry where it does not.\n\n    The tag is present on exactly half the studies here and is sometimes an empty\n    string rather than absent, which is not the same as NaN. Treating the other half\n    as left-sided is what `normalise_laterality` did by omission, so the geometry\n    fallback is not a refinement - it is the difference between normalising half the\n    corpus and normalising all of it.\n    \"\"\"\n    geo = side_from_corner_x(h) if RULES[\"lat\"] == \"corner_x\" else side_from_geometry(h)\n    d, n_tag, n_geo, n_none, n_disagree = {}, 0, 0, 0, 0\n    for st, g in h.groupby(\"StudyInstanceUID\"):\n        v = [str(x).strip().upper() for x in g[\"Laterality\"].dropna()]\n        if RULES[\"lat\"] == \"corner_x\" and \"ImageLaterality\" in g.columns:\n            # The legacy rule reads the second tag too, so a study tagged only there is\n            # resolved from the tag rather than from geometry.\n            v += [str(x).strip().upper() for x in g[\"ImageLaterality\"].dropna()]\n        v = [x[0] for x in v if x and x[0] in (\"L\", \"R\")]\n        side = v[0] if v else None\n        if side is not None:\n            n_tag += 1\n            if geo.get(st) is not None and geo[st] != side:\n                n_disagree += 1\n        else:\n            side = geo.get(st)\n            n_geo += side is not None\n            n_none += side is None\n        d[st] = side\n    log(f\"{tag}laterality: {n_tag} from the tag, {n_geo} from geometry, \"\n        f\"{n_none} unresolved; tag and geometry disagree on {n_disagree} \"\n        f\"({n_disagree / max(n_tag, 1):.1%} of the tagged)\")\n    return d\n\n\n\ndef probe(item):\n    split, study, series, path = item\n    row = {\"split\": split, \"StudyInstanceUID\": study, \"SeriesInstanceUID\": series,\n           \"dir\": path}\n    try:\n        files = sorted(e.name for e in os.scandir(path) if e.name.endswith(\".dcm\"))\n        row[\"files\"] = files\n        row[\"n_slices\"] = len(files)\n        if not files:\n            return row\n        ds = pydicom.dcmread(os.path.join(path, files[len(files) // 2]),\n                             stop_before_pixels=True, force=True)\n        for t in HDR_TAGS:\n            v = getattr(ds, t, None)\n            if v is None:\n                row[t] = None\n            elif isinstance(v, (list, tuple)) or type(v).__name__ == \"MultiValue\":\n                row[t] = \"|\".join(str(x) for x in v)\n            else:\n                row[t] = str(v)\n    except Exception as exc:\n        row[\"err\"] = str(exc)[:120]\n    return row\n\n\ndef walk(split):\n    \"\"\"Every series directory of a split, with one header read per series.\n\n    An absent split returns an empty frame *with the columns annotate expects*. Returning\n    a bare DataFrame looks like the same thing and is not: the next call indexes\n    `SeriesDescription` and raises KeyError, so the branch that exists to survive a\n    missing split is what turns it into a crash.\n    \"\"\"\n    base = ROOT / split\n    items = []\n    if not base.is_dir():\n        return pd.DataFrame(columns=[\"split\", \"StudyInstanceUID\", \"SeriesInstanceUID\",\n                                     \"dir\", \"files\", \"n_slices\"] + HDR_TAGS)\n    for study in os.scandir(base):\n        if study.is_dir():\n            for series in os.scandir(study.path):\n                if series.is_dir():\n                    items.append((split, study.name, series.name, series.path))\n    with ThreadPoolExecutor(max_workers=HDR_THREADS) as pool:\n        rows = list(pool.map(probe, items))\n    return pd.DataFrame(rows)\n\n\ndef annotate(df):\n    \"\"\"Recover fat suppression and pulse-sequence weighting from the header.\"\"\"\n    desc = (df[\"SeriesDescription\"].fillna(\"\") + \" \" + df[\"SequenceName\"].fillna(\"\"))\n    desc = desc.str.lower().str.replace(_SEP, \" \", regex=True)\n\n    opts = df[\"ScanOptions\"].fillna(\"\").str.upper().str.split(\"|\")\n    # GE writes SAT_GEMS for spatial saturation, so ScanOptions must be matched as\n    # exact tokens; a substring test on \"SAT\" fires on non-fat-sat series.\n    opts_fs = opts.apply(lambda ts: any(t.strip() in FATSAT_OPTS for t in ts))\n    df[\"fatsat\"] = desc.str.contains(_FATSAT_RX) | opts_fs\n\n    tr = pd.to_numeric(df[\"RepetitionTime\"], errors=\"coerce\")\n    te = pd.to_numeric(df[\"EchoTime\"], errors=\"coerce\")\n    gre = df[\"ScanningSequence\"].fillna(\"\").str.upper().str.contains(\"GR\")\n    t1, t2, pdw = desc.str.contains(_T1_RX), desc.str.contains(_T2_RX), desc.str.contains(_PD_RX)\n\n    df[\"weight\"] = np.where(t1 & ~t2 & ~pdw, \"T1\",\n                     np.where(t2 & ~pdw, \"T2\",\n                       np.where(pdw, \"PD\",\n                         np.where(gre, \"GRE\",\n                           np.where(tr < 800, \"T1\",\n                             np.where(te > 60, \"T2\",\n                               np.where(tr >= 800, \"PD\", \"UNK\")))))))\n    df[\"fluid\"] = np.isin(df[\"weight\"], [\"PD\", \"T2\"])\n    df[\"px\"] = pd.to_numeric(\n        df[\"PixelSpacing\"].fillna(\"\").str.split(\"|\").str[0].replace(\"\", np.nan),\n        errors=\"coerce\")\n    return df\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-10T06:00:15.836868Z","iopub.status.busy":"2026-08-10T06:00:15.836126Z","iopub.status.idle":"2026-08-10T06:00:15.859358Z","shell.execute_reply":"2026-08-10T06:00:15.858534Z"},"papermill":{"duration":0.055198,"end_time":"2026-08-10T06:00:15.860772+00:00","exception":false,"start_time":"2026-08-10T06:00:15.805574+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"52aad610","cell_type":"code","source":"def pick_slots(series_df, plane_map):\n    \"\"\"One series per slot per study.\n\n    Ties are broken toward the stack with the most slices: a thicker stack samples the\n    joint more densely, and the three-slice sampler below benefits from the margin.\n    \"\"\"\n    series_df = series_df.copy()\n    series_df[\"plane\"] = series_df[\"SeriesInstanceUID\"].map(plane_map)\n    out = {}\n    for study, g in series_df.groupby(\"StudyInstanceUID\"):\n        chosen = {}\n        for name, plane, fluid, fs in SLOTS:\n            sel = (g[\"plane\"] == plane) & (g[\"fatsat\"] == fs)\n            # fluid=None means \"do not condition on weighting\" - the public scheme,\n            # where the single provided flag stands in for both axes at once.\n            if fluid is not None:\n                sel &= (g[\"fluid\"] == fluid)\n            cand = g[sel]\n            # A slot with no series matching its predicate stays empty, and no substitute\n            # is admitted from a neighbouring predicate. Relaxing the weighting to fill a\n            # T1 slot would draw from the pool `SAG_FLUID_NOFS` selects from, since that\n            # pool is what remains once the weighting is dropped: over the training corpus\n            # it would put one series in two slots for 2383 of 4407 studies and leave 56%\n            # of the T1 slot holding PD or T2. The presence mask would then assert a\n            # sequence that was never acquired, and the per-diagnosis softmax of §6 would\n            # divide its attention across two identical slots, giving one acquisition\n            # about twice the weight it carries in a study that holds both. The mask is\n            # there to say a slot is absent, which is what an absent slot is.\n            if len(cand) == 0 and RULES[\"slot_fallback\"] and fluid is False:\n                # The relaxation the paragraph above rejects, reproduced because an\n                # imported member was fitted with its T1 slots filled this way: over half\n                # of that member's training studies had a T1 slot holding a series that\n                # is not T1. Leaving those slots empty would present it with a presence\n                # mask it never saw.\n                cand = g[(g[\"plane\"] == plane) & (~g[\"fatsat\"])]\n            if len(cand):\n                chosen[name] = cand.sort_values(\"n_slices\", ascending=False).iloc[0]\n        out[study] = chosen\n    return out\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-10T06:00:15.900763Z","iopub.status.busy":"2026-08-10T06:00:15.90042Z","iopub.status.idle":"2026-08-10T06:00:15.907001Z","shell.execute_reply":"2026-08-10T06:00:15.906258Z"},"papermill":{"duration":0.029057,"end_time":"2026-08-10T06:00:15.908764+00:00","exception":false,"start_time":"2026-08-10T06:00:15.879707+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"62bc0ae2","cell_type":"code","source":"ORDER_TAGS = [(0x0020, 0x0032), (0x0020, 0x0037), (0x0020, 0x0013)]\n\n# Series in which at least one sampled slice would not decode. A list rather than a\n# counter because appending is atomic under the reader threads, and reported rather than\n# swallowed: unreported, a decode failure is indistinguishable from a black knee.\nDECODE_FAILED = []\n\n\ndef cache_tag(rules=None):\n    \"\"\"The name a decoded cache is stored under.\n\n    It has to name everything that decides the pixels, not only their dimensions. Two\n    configurations that agree on resolution, slice count, crop and band but disagree on\n    how a slice is chosen produce different arrays of identical shape - so a tag built\n    from the dimensions alone lets the second attach to the first one's file and train\n    against pixels it never asked for, with nothing anywhere reporting a mismatch.\n\n    A native reading keeps the plain name, so caches decoded before the rules existed\n    stay valid; anything else earns a suffix.\n    \"\"\"\n    r = dict(RULES if rules is None else rules)\n    t = (f\"{CACHE_IMG}px_{CACHE_SLICES}sl_{int(CROP_MM)}mm_\"\n         f\"{SLICE_BAND[0]:.2f}-{SLICE_BAND[1]:.2f}\")\n    if {k: r.get(k, v) for k, v in RULES_NATIVE.items()} != RULES_NATIVE:\n        t += \"_\" + hashlib.md5(json.dumps(r, sort_keys=True).encode()).hexdigest()[:6]\n    return t\n\n\ndef _natural_key(name):\n    return tuple(int(x) if x.isdigit() else x.lower()\n                 for x in re.split(r\"(\\d+)\", str(name)))\n\n\ndef _order_dominant_axis(rec):\n    \"\"\"The slice order an imported member was fitted under.\n\n    It sorts on the raw patient coordinate along whichever axis varies most across the\n    stack, rather than on the projection onto the slice normal. The two differ by a sign,\n    not by a formula: measured over this corpus every sagittal series has a slice normal\n    with n_x in [-1.00, -0.98], so p.n is the negative of the raw x this sorts on and the\n    two stacks come out exactly reversed. Because the band sampler truncates rather than\n    rounds, its nine indices are not symmetric about the middle, so nine slices drawn from\n    a twenty-six slice stack under one order share two with the other.\n\n    Missing geometry falls back to `InstanceNumber` and then to a natural sort of the file\n    name, both at the same 80% threshold the imported pipeline used.\n    \"\"\"\n    files, d = rec[\"files\"], rec[\"dir\"]\n    rows = []\n    for pos, f in enumerate(files):\n        ipp = inst = None\n        try:\n            ds = pydicom.dcmread(os.path.join(d, f), force=True, stop_before_pixels=True,\n                                 specific_tags=[\"ImagePositionPatient\", \"InstanceNumber\"])\n            raw = getattr(ds, \"ImagePositionPatient\", None)\n            if raw is not None and len(raw) >= 3:\n                c = np.asarray(raw[:3], dtype=np.float64)\n                if np.isfinite(c).all():\n                    ipp = c\n            n = getattr(ds, \"InstanceNumber\", None)\n            if n is not None:\n                inst = float(n)\n        except Exception:\n            pass\n        rows.append((f, ipp, inst, pos))\n\n    placed = [r for r in rows if r[1] is not None]\n    need = max(2, int(0.8 * len(rows)))\n    if len(placed) >= need:\n        xyz = np.stack([r[1] for r in placed])\n        axis = int(np.argmax(np.ptp(xyz, axis=0)))\n        spare = float(np.nanmedian(xyz[:, axis]))\n        rows.sort(key=lambda r: (float(r[1][axis]) if r[1] is not None else spare,\n                                 r[2] if r[2] is not None else float(\"inf\"), r[3]))\n    elif sum(r[2] is not None for r in rows) >= need:\n        rows.sort(key=lambda r: (r[2] if r[2] is not None else float(\"inf\"), r[3]))\n    else:\n        rows.sort(key=lambda r: _natural_key(r[0]))\n    return [r[0] for r in rows], True\n\n\ndef order_slices(rec):\n    \"\"\"Return the series' files sorted along the through-plane axis.\n\n    A DICOM file name here is a SOP Instance UID, which is assigned arbitrarily. Sorting\n    by it therefore produces an order uncorrelated with anatomy - measured over one\n    series, Spearman between file-name rank and physical position is 0.009, i.e. none.\n    Anything that assumes the file order means something is then operating on noise: the\n    three channels of a \"2.5D\" input are three unrelated views rather than neighbouring\n    slices, \"the middle of the stack\" is a random subset, and reversing slice order to\n    normalise laterality reverses nothing meaningful.\n\n    The physical order is recoverable exactly. Each slice carries its position in patient\n    coordinates and the in-plane axes; projecting the position onto the slice normal\n    gives a signed through-plane coordinate, monotonic along the stack:\n\n        n = r_x  x  r_y ,      k = p . n\n\n    `InstanceNumber` is the fallback. It usually tracks the projection up to sign, but\n    interleaved and multi-echo acquisitions need not number slices in the order they\n    occupy in space - but the projection is signed in patient\n    coordinates, which is what laterality normalisation needs.\n    \"\"\"\n    if RULES[\"order\"] == \"dominant_axis\":\n        return _order_dominant_axis(rec)\n    files, d = rec[\"files\"], rec[\"dir\"]\n    keyed = []\n    for f in files:\n        k = None\n        try:\n            ds = pydicom.dcmread(os.path.join(d, f), force=True, stop_before_pixels=True,\n                                 specific_tags=ORDER_TAGS)\n            iop = np.asarray(ds.ImageOrientationPatient, dtype=float)\n            ipp = np.asarray(ds.ImagePositionPatient, dtype=float)\n            k = float(np.dot(ipp, np.cross(iop[:3], iop[3:])))\n        except Exception:\n            try:\n                k = float(ds.InstanceNumber)\n            except Exception:\n                k = None\n        keyed.append((k, f))\n    if any(k is None for k, _ in keyed):\n        # A series with no usable geometry keeps its arbitrary order; that is worse than\n        # sorting but better than dropping the series, and it is logged as a count.\n        return files, False\n    return [f for _, f in sorted(keyed, key=lambda t: t[0])], True\n\n\ndef read_slot(rec, n_slice=None, out_size=None):\n    \"\"\"`n_slice` physically spread slices from one series, at `out_size` pixels.\n\n    Returns uint8 [n_slice, out, out] normalised per-series to its 1st-99th\n    percentile. Percentiles rather than min/max because MR intensity has no absolute\n    scale and a single bright vessel would otherwise compress the whole dynamic range.\n\n    Reading is the expensive half of this pipeline, so the caller reads once at the\n    largest configuration it needs and derives the smaller ones from the returned buffer\n    rather than re-reading.\n    \"\"\"\n    n_slice = GROUP if n_slice is None else n_slice\n    out_size = IMG if out_size is None else out_size\n    files, d, px = rec.get(\"ordered\") or rec[\"files\"], rec[\"dir\"], rec[\"px\"]\n    n = len(files)\n    if n == 0:\n        return None\n    # Spread the samples over a central band of the stack: the outermost slices of a knee\n    # series are mostly soft tissue outside the joint. The band is a constant rather than\n    # a literal because how much of the stack is worth reading depends on how many slices\n    # are being taken - at three the middle is all that fits, while at sixteen the ends\n    # are worth having, and a Baker cyst sits at the posteromedial end of a sagittal one.\n    lo, hi = int(SLICE_BAND[0] * (n - 1)), int(SLICE_BAND[1] * (n - 1))\n    idx = np.unique(np.linspace(lo, hi, n_slice).astype(int)) if hi > lo else np.array([n // 2])\n    while len(idx) < n_slice:\n        idx = np.append(idx, idx[-1])\n\n    planes = []\n    for i in idx[:n_slice]:\n        try:\n            ds = pydicom.dcmread(os.path.join(d, files[int(i)]), force=True)\n            a = ds.pixel_array.astype(np.float32)\n            sl = float(getattr(ds, \"RescaleSlope\", 1) or 1)\n            ic = float(getattr(ds, \"RescaleIntercept\", 0) or 0)\n            a = a * sl + ic\n        except Exception:\n            a = None                      # no shape is known here; see below\n        planes.append(a)\n\n    # A slice that would not decode has no shape of its own, and inventing one is how a\n    # single unreadable file erases a whole series: a substitute allocated at the resize\n    # target while the decoded slices are still native makes the shape check below take\n    # the substitute as the authority and zero the good slices with it, leaving a black\n    # slot that the presence mask still reports as acquired.\n    #\n    # A failure is instead filled from the nearest slice that did decode - the same\n    # convention the sampler already uses when the band holds fewer distinct slices than\n    # were asked for - and a series where nothing decodes is reported absent, which the\n    # mask can express, rather than black, which it cannot.\n    got = [k for k, p in enumerate(planes) if p is not None]\n    if RULES[\"decode_fill\"] == \"zero\":\n        # What an imported member was fitted with: a failure becomes a zero plane at the\n        # resize target, which the shape check below then propagates to the whole slot.\n        # It is the behaviour the paragraph above describes and rejects, kept here only\n        # because that member's weights were learned against slots blacked out this way.\n        if not got:\n            DECODE_FAILED.append(rec.get(\"SeriesInstanceUID\", d))\n        planes = [np.zeros((out_size, out_size), np.float32) if p is None else p\n                  for p in planes]\n        got = list(range(len(planes)))\n    if not got:\n        DECODE_FAILED.append(rec.get(\"SeriesInstanceUID\", d))\n        return None\n    if len(got) < len(planes):\n        DECODE_FAILED.append(rec.get(\"SeriesInstanceUID\", d))\n        for k, p in enumerate(planes):\n            if p is None:\n                planes[k] = planes[min(got, key=lambda j: abs(j - k))]\n\n    # Slices of one series can still differ in matrix size - multi-echo and some\n    # reformats do - and those are genuinely not stackable.\n    shp = planes[0].shape\n    planes = [p if p.shape == shp else np.zeros(shp, np.float32) for p in planes]\n    vol = np.stack(planes)\n\n    # constant physical extent, then resize: PixelSpacing varies 3.4x across the corpus\n    if px and np.isfinite(px) and px > 0:\n        want = int(round(CROP_MM / px))\n        h, w = shp\n        if 16 < want < min(h, w):\n            cy, cx = h // 2, w // 2\n            half = want // 2\n            vol = vol[:, max(0, cy - half):cy + half, max(0, cx - half):cx + half]\n\n    lo_v, hi_v = np.percentile(vol, [1, 99])\n    vol = np.clip((vol - lo_v) / max(hi_v - lo_v, 1e-6), 0, 1)\n\n    t = torch.from_numpy(np.ascontiguousarray(vol)).unsqueeze(0)\n    t = F.interpolate(t, size=(out_size, out_size), mode=\"bilinear\", align_corners=False)\n    # uint8, not float32. These buffers queue up between the reader threads and the\n    # encoder, and at this size a float32 slot-series is several megabytes. Intensity is\n    # already normalised into [0, 1] here, so eight bits cost nothing that a bilinear\n    # resize has not already cost, and the queue is a quarter the size.\n    return (t.squeeze(0) * 255).round().clamp(0, 255).to(torch.uint8)\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-10T06:00:16.024426Z","iopub.status.busy":"2026-08-10T06:00:16.024194Z","iopub.status.idle":"2026-08-10T06:00:16.04897Z","shell.execute_reply":"2026-08-10T06:00:16.048071Z"},"papermill":{"duration":0.046869,"end_time":"2026-08-10T06:00:16.050509+00:00","exception":false,"start_time":"2026-08-10T06:00:16.00364+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"7f3af782","cell_type":"code","source":"def normalise_laterality(img, plane, lat):\n    \"\"\"Map every knee onto a left-knee convention.\n\n    Coronal and axial views mirror under a horizontal flip. Sagittal stacks are not\n    mirror images of each other - the slice order runs medial-to-lateral in opposite\n    directions - so the channel order is reversed instead.\n    \"\"\"\n    if lat != \"R\":\n        return img\n    if plane in (\"Coronal\", \"Axial\"):\n        return torch.flip(img, dims=[-1])\n    return torch.flip(img, dims=[0])\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-10T06:00:16.13013Z","iopub.status.busy":"2026-08-10T06:00:16.129434Z","iopub.status.idle":"2026-08-10T06:00:16.134079Z","shell.execute_reply":"2026-08-10T06:00:16.133318Z"},"papermill":{"duration":0.025651,"end_time":"2026-08-10T06:00:16.135389+00:00","exception":false,"start_time":"2026-08-10T06:00:16.109738+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"4f6bb7cf","cell_type":"code","source":"# Where the geometric slice order may be remembered between runs. Unset on the platform,\n# because each run gets a fresh machine and there is nothing to remember; set off it,\n# where the same corpus is cached again at every resolution and slice count and the order\n# is a function of neither. It is opt-in so that the scored run's behaviour is decided by\n# the code rather than by whether a file happens to be lying about.\nORDER_CACHE = os.environ.get(\"RSNA_ORDER_CACHE\") or None\n\n\ndef build_cache(slot_map, plane_map, lat_map, tag):\n    \"\"\"Decode every (study, slot) once into an in-memory uint8 array.\n\n    Fine-tuning revisits the same pixels every epoch. Reading them from the mount each\n    time would make the epoch count a function of I/O rather than of learning, so they\n    are decoded once and held as bytes: intensity has already been normalised into\n    [0, 1], and eight bits cost nothing a bilinear resize has not already cost.\n\n    CACHE_SLICES positions are kept per slot, which the training loop reads as N_GROUP\n    groups of GROUP consecutive channels.\n    \"\"\"\n    studies = sorted(slot_map)\n    sidx = {s: i for i, s in enumerate(studies)}\n    cache = np.zeros((len(studies), N_SLOT, CACHE_SLICES, IMG, IMG), np.uint8)\n    mask = np.zeros((len(studies), N_SLOT), np.float32)\n    log(f\"{tag}: cache {cache.shape} = {cache.nbytes / 1024 ** 3:.1f} GB\")\n\n    jobs = [(st, k, plane, slot_map[st][name])\n            for st in studies\n            for k, (name, plane, _, _) in enumerate(SLOTS)\n            if name in slot_map[st]]\n    n_job = len(jobs)\n\n    # Ordering first, and as its own pass. It reads one header per slice of every chosen\n    # series - far more file opens than the decode that follows - and on a network mount\n    # that is latency, not work, so it gets its own wider pool.\n    t_ord = time.time()\n    n_slice_total = sum(len(j[3][\"files\"]) for j in jobs)\n    log(f\"{tag}: ordering {len(jobs)} slot-series ({n_slice_total} slice headers)\")\n    ok = done = 0\n    CHUNK_O = 1024\n\n    # A remembered order, when one is offered. The projection depends on the DICOM\n    # geometry alone, so it is the same at every resolution and every slice count, and\n    # it costs one header read per slice - the largest single cost in this pass. An entry\n    # is validated by the number of files present, so a tree that has changed under it is\n    # recomputed rather than trusted: order is derived data, and a stale entry would be\n    # invisible in the way that matters most.\n    seen = {}\n    if ORDER_CACHE and Path(ORDER_CACHE).is_file():\n        try:\n            import json as _json\n            seen = _json.loads(Path(ORDER_CACHE).read_text())\n        except (OSError, ValueError):\n            seen = {}\n        hit = 0\n        for _, _, _, rec in jobs:\n            e = seen.get(rec[\"SeriesInstanceUID\"])\n            if e and len(e[\"files\"]) == len(rec[\"files\"]):\n                rec[\"ordered\"] = e[\"files\"]\n                ok += int(e[\"good\"])\n                hit += 1\n        jobs = [j for j in jobs if \"ordered\" not in j[3]]\n        log(f\"{tag}: {hit} slot-series ordered from {ORDER_CACHE}, {len(jobs)} to read\")\n\n    with ThreadPoolExecutor(max_workers=ORDER_THREADS) as pool:\n        for c0 in range(0, len(jobs), CHUNK_O):\n            block = jobs[c0:c0 + CHUNK_O]\n            for (_, _, _, rec), (files, good) in zip(\n                    block, pool.map(lambda j: order_slices(j[3]), block)):\n                rec[\"ordered\"] = files\n                ok += int(good)\n                done += 1\n                if ORDER_CACHE:\n                    seen[rec[\"SeriesInstanceUID\"]] = {\"files\": files, \"good\": bool(good)}\n            # The ceiling is whichever comes first: the pass's own budget, or the share\n            # of what is left of the run that it may take. The second is what makes the\n            # first safe to set generously - a mount slow enough to matter cannot spend\n            # the training time, because the budget shrinks as the run does.\n            budget = min(ORDER_BUDGET_S, max(60.0, (TIME_BUDGET - (time.time() - T0)) * 0.35))\n            if time.time() - t_ord > budget:\n                log(f\"{tag}: ordering budget spent at {done}/{len(jobs)}; \"\n                    f\"the rest keep file order\")\n                break\n    if ORDER_CACHE and done:\n        import json as _json\n        _t = Path(ORDER_CACHE).with_suffix(\".tmp\")\n        _t.write_text(_json.dumps(seen))\n        _t.replace(Path(ORDER_CACHE))\n    log(f\"{tag}: ordered {ok}/{n_job} by geometry \"\n        f\"({n_job - ok} kept arbitrary) in {time.time() - t_ord:.0f}s\")\n\n    jobs = [(st, k, plane, slot_map[st][name])\n            for st in studies\n            for k, (name, plane, _, _) in enumerate(SLOTS)\n            if name in slot_map[st]]\n    log(f\"{tag}: decoding {len(jobs)} slot-series\")\n    n_failed_before = len(DECODE_FAILED)\n\n    CHUNK = 512\n    done = 0\n    with ThreadPoolExecutor(max_workers=PIX_THREADS) as pool:\n        for c0 in range(0, len(jobs), CHUNK):\n            block = jobs[c0:c0 + CHUNK]\n            for (st, k, plane, _), img in zip(\n                    block, pool.map(lambda j: read_slot(j[3], CACHE_SLICES, IMG), block)):\n                done += 1\n                if img is None:\n                    continue\n                cache[sidx[st], k] = normalise_laterality(img, plane,\n                                                          lat_map.get(st)).numpy()\n                mask[sidx[st], k] = 1.0\n            if done % 4096 < CHUNK:\n                log(f\"  {tag} {done}/{len(jobs)}\")\n            if time.time() - T0 > TIME_BUDGET:\n                log(f\"  {tag}: time budget reached during decode\")\n                break\n    n_failed = len(DECODE_FAILED) - n_failed_before\n    log(f\"{tag}: {int(mask.sum())}/{len(jobs)} slots filled\"\n        + (f\"; {n_failed} series had a slice that would not decode\" if n_failed else \"\"))\n    gc.collect()\n    return studies, cache, mask\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-10T06:00:16.213041Z","iopub.status.busy":"2026-08-10T06:00:16.212611Z","iopub.status.idle":"2026-08-10T06:00:16.228002Z","shell.execute_reply":"2026-08-10T06:00:16.227303Z"},"papermill":{"duration":0.037051,"end_time":"2026-08-10T06:00:16.229252+00:00","exception":false,"start_time":"2026-08-10T06:00:16.192201+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"9ca87656","cell_type":"code","source":"class SlotHead(nn.Module):\n    \"\"\"Per-diagnosis attention over the slot embeddings of one study.\n\n    Each finding is read on particular sequences - cruciates sagittally, collateral\n    ligaments and the meniscal body coronally, patellar cartilage axially - so pooling\n    the slots identically would dilute the one that carries the evidence with the rest.\n\n    The aggregation is deliberately this simple. With a study-level label there is no\n    signal telling the model which part of a study matters, so extra attention\n    parameters below the slot level would have nothing to learn from and would spend\n    their capacity fitting noise.\n    \"\"\"\n\n    def __init__(self, dim, n_slot, n_out, hidden=256, p=0.2, prior=False):\n        super().__init__()\n        self.proj = nn.Sequential(nn.LayerNorm(dim), nn.Linear(dim, hidden), nn.GELU())\n        self.slot_emb = nn.Parameter(torch.randn(n_slot, hidden) * 0.02)\n        self.query = nn.Parameter(torch.randn(n_out, hidden) * 0.02)\n        self.drop = nn.Dropout(p)\n        self.out = nn.Linear(hidden, n_out)\n        self.hidden = hidden\n        # An imported member carries a fixed per-(diagnosis, slot) tilt on the attention\n        # logits, set from the anatomy table below rather than learned. It is a buffer, so\n        # it travels in the state dict and must exist for that member to load; exp(0.55)\n        # gives a preferred slot about 1.73x the weight of an unpreferred one, which\n        # biases the softmax without ever excluding a slot.\n        p_ = torch.zeros(n_out, n_slot)\n        if prior and n_slot == len(SLOTS) and n_out == len(TARGETS):\n            for t, slots in SLOT_PRIOR_TABLE.items():\n                if t in TARGETS:\n                    p_[TARGETS.index(t), list(slots)] = SLOT_PRIOR_STRENGTH\n        self.prior = prior\n        if prior:\n            self.register_buffer(\"slot_prior\", p_)\n\n    def forward(self, x, mask):\n        h = self.proj(x) + self.slot_emb\n        att = torch.einsum(\"bsh,oh->bos\", h, self.query) / self.hidden ** 0.5\n        if self.prior:\n            att = att + self.slot_prior.unsqueeze(0)\n        att = att.masked_fill(mask.unsqueeze(1) < 0.5, -1e4).softmax(-1)\n        ctx = self.drop(torch.einsum(\"bos,bsh->boh\", att, h))\n        return (ctx * self.out.weight.unsqueeze(0)).sum(-1) + self.out.bias\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-10T06:00:16.309396Z","iopub.status.busy":"2026-08-10T06:00:16.308893Z","iopub.status.idle":"2026-08-10T06:00:16.317017Z","shell.execute_reply":"2026-08-10T06:00:16.316238Z"},"papermill":{"duration":0.029836,"end_time":"2026-08-10T06:00:16.318389+00:00","exception":false,"start_time":"2026-08-10T06:00:16.288553+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"1dcecf7f","cell_type":"code","source":"class Model(nn.Module):\n    \"\"\"Encoder plus head, trained end to end.\n\n    A study arrives as a bag of slot images. The bag is flattened for the encoder and\n    folded back before the head, so the encoder never sees the study structure and the\n    head never sees pixels.\n    \"\"\"\n\n    def __init__(self, backbone, dim, pool=\"cls_mean\", prior=False):\n        super().__init__()\n        self.backbone = backbone\n        self.pool = pool\n        self.head = SlotHead(dim * POOL_PARTS[pool], N_SLOT, len(TARGETS), prior=prior)\n        self.register_buffer(\"mean\", torch.tensor([0.485, 0.456, 0.406]).view(1, 3, 1, 1))\n        self.register_buffer(\"std\", torch.tensor([0.229, 0.224, 0.225]).view(1, 3, 1, 1))\n\n    def forward(self, imgs, mask, img_size=None):\n        B, S = imgs.shape[:2]\n        x = imgs.reshape(B * S, *imgs.shape[2:]).float().div_(255.0)\n        if img_size is not None and img_size != x.shape[-1]:\n            # The cache is held at the highest resolution any configuration needs; the\n            # rest downsample from it, so every configuration sees the same pixels\n            # through a different sampling grid rather than a different crop.\n            x = F.interpolate(x, size=(img_size, img_size), mode=\"bilinear\",\n                              align_corners=False)\n        x = (x - self.mean) / self.std\n        out = self.backbone(pixel_values=x).last_hidden_state\n        patch = out[:, 1:]\n        parts = [out[:, 0], patch.mean(1)]\n        if self.pool == \"cls_mean_focal\":\n            # The upper tail of each channel over the patch grid, taken per channel\n            # rather than by selecting whole patches: a finding occupies a small part of\n            # the field, so a plain mean over 256 patches dilutes it by two orders of\n            # magnitude, and this keeps the top eighth of each channel's responses.\n            k = max(1, patch.shape[1] // 8)\n            parts.append(patch.topk(k, dim=1).values.mean(1))\n        feat = torch.cat(parts, dim=1).reshape(B, S, -1)\n        return self.head(feat, mask)\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-10T06:00:16.358463Z","iopub.status.busy":"2026-08-10T06:00:16.357913Z","iopub.status.idle":"2026-08-10T06:00:16.366159Z","shell.execute_reply":"2026-08-10T06:00:16.365609Z"},"papermill":{"duration":0.030144,"end_time":"2026-08-10T06:00:16.367434+00:00","exception":false,"start_time":"2026-08-10T06:00:16.33729+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"919634b2","cell_type":"code","source":"def build_model(unfreeze_last, source=None, variant=\"small\", pool=\"cls_mean\",\n                prior=False):\n    \"\"\"Load the encoder and open the last `unfreeze_last` blocks for training.\n\n    The early blocks of a self-supervised transformer are generic edge and texture\n    filters; the late blocks carry semantics. Opening only the late ones is the cautious\n    choice - there may not be enough supervision here to improve the early ones and there\n    is certainly enough to damage them - but how far the line should sit is a question\n    the corpus has to answer rather than the intuition.\n\n    `source` names where the weights come from. Left unset it is the attached model\n    directory, which is the only thing available here. It is a parameter so that a run\n    off the platform builds the same object from the same code rather than from a second\n    definition that has to be kept in step by hand.\n    \"\"\"\n    from transformers import AutoModel\n    p = source if source is not None else find_dinov2(variant)\n    if p is None:\n        raise FileNotFoundError(\"DINOv2 weights not attached\")\n    bb = AutoModel.from_pretrained(str(p))\n    n_layer = len(bb.encoder.layer)\n    for prm in bb.parameters():\n        prm.requires_grad = False\n    for blk in bb.encoder.layer[max(0, n_layer - unfreeze_last):]:\n        for prm in blk.parameters():\n            prm.requires_grad = True\n    for prm in bb.layernorm.parameters():\n        prm.requires_grad = True\n    dim = bb.config.hidden_size\n    trainable = sum(p.numel() for p in bb.parameters() if p.requires_grad)\n    log(f\"backbone: {n_layer} blocks, last {unfreeze_last} trainable \"\n        f\"({trainable / 1e6:.1f}M params), feature dim {dim * POOL_PARTS[pool]}\")\n    return Model(bb, dim, pool=pool, prior=prior)\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-10T06:00:16.445842Z","iopub.status.busy":"2026-08-10T06:00:16.44554Z","iopub.status.idle":"2026-08-10T06:00:16.451484Z","shell.execute_reply":"2026-08-10T06:00:16.450902Z"},"papermill":{"duration":0.027474,"end_time":"2026-08-10T06:00:16.453046+00:00","exception":false,"start_time":"2026-08-10T06:00:16.425572+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"52ffca9a","cell_type":"code","source":"FINGERPRINT_TOL = 2e-3\n\n\ndef fingerprint(model, dev, img_size, n_slot=None, group=None, seed=None):\n    \"\"\"The model's output on a fixed synthetic bag, as a portable identity.\n\n    Weights that are loaded but read through the wrong preprocessing produce predictions,\n    not errors. The submission is well formed, the log says nothing, and the difference is\n    a number no output of the run reveals. Scaling that never happens, or happens twice,\n    is enough on its own and changes no shape anywhere.\n\n    So a set of weights carries the answer it gave to a question with no data in it. The\n    input is generated from a seed rather than read, so it is the same on any machine, and\n    it is pushed through the whole forward path - the byte scaling, the ImageNet\n    normalisation, the resize, the encoder, the slot attention. Any of those differing\n    moves the output by order one. Numerics differing between two GPUs moves it by about\n    1e-5, which is why the tolerance sits between them rather than at zero.\n\n    This checks that the model computes what it computed when it was fitted. It cannot\n    check that the pixels reaching it are the right pixels; `read_slot` and the header\n    pass answer to their own tests.\n    \"\"\"\n    n_slot = N_SLOT if n_slot is None else n_slot\n    group = GROUP if group is None else group\n    seed = SEED if seed is None else seed\n    g = torch.Generator().manual_seed(seed)\n    imgs = torch.randint(0, 256, (2, n_slot, group, img_size, img_size),\n                         generator=g, dtype=torch.uint8).to(dev)\n    mask = torch.ones(2, n_slot, device=dev)\n    mask[1, -1] = 0.0                       # exercise the masked branch of the softmax\n    was_training = model.training\n    model.eval()\n    with torch.no_grad():\n        # float32 throughout: autocast would make the value depend on which device\n        # happened to run it, and the point of the number is that it does not.\n        out = model(imgs, mask, img_size).float().cpu().numpy()\n    if was_training:\n        model.train()\n    return out\n\n\ndef check_fingerprint(model, dev, img_size, expected, tol=FINGERPRINT_TOL, tag=\"\"):\n    \"\"\"Compare against a stored fingerprint; raise when the model is not the same map.\"\"\"\n    got = fingerprint(model, dev, img_size)\n    exp = np.asarray(expected, np.float32)\n    if got.shape != exp.shape:\n        raise WeightsError(f\"{tag}fingerprint shape {got.shape} != stored {exp.shape}: \"\n                           f\"the architecture is not the one these weights were fitted to\")\n    d = float(np.abs(got - exp).max())\n    if d > tol:\n        raise WeightsError(\n            f\"{tag}fingerprint differs by {d:.4g} (tolerance {tol:g}). The weights load \"\n            f\"but do not compute what they computed when fitted - preprocessing, \"\n            f\"resolution or architecture has moved between the two runs.\")\n    log(f\"{tag}fingerprint matches within {d:.2g}\")\n    return d\n\n\nclass WeightsError(RuntimeError):\n    \"\"\"Raised when attached weights cannot be trusted to be the ones that were fitted.\n\n    Deliberately fatal for the same reason as LabelSourceError: a run that predicts from\n    a mismatched model completes, writes a plausible submission, and differs from a\n    correct one only in a number no output of the run reveals.\n    \"\"\"\n\n\ndef find_weights(name=\"manifest.json\"):\n    \"\"\"Locate a mounted weights package, or return None if none is attached.\n\n    Same shape as `find_label_table`: the notebook must keep working for a reader who\n    attaches nothing, so absence is a path rather than an error. What must not be silent\n    is a package that is attached and unusable, and that is what `load_weights` refuses.\n    \"\"\"\n    import json\n    base = Path(\"/kaggle/input\")\n    if not base.is_dir():\n        return None\n    for root, dirs, files in os.walk(base):\n        dirs[:] = [d for d in dirs if d not in (\"train_series\", \"test_series\")]\n        if name not in files:\n            continue\n        # The manifest decides, not the filenames beside it. Testing for a naming\n        # convention makes the search agree with whatever the packager happened to call\n        # its files last, which is a second definition of what a package is.\n        try:\n            man = json.loads((Path(root) / name).read_text())\n        except (OSError, ValueError):\n            continue\n        if isinstance(man.get(\"members\"), list) and man[\"members\"]:\n            missing = [m[\"file\"] for m in man[\"members\"]\n                       if not (Path(root) / m[\"file\"]).is_file()]\n            if missing:\n                raise WeightsError(\n                    f\"{root} holds a manifest listing {len(man['members'])} members but \"\n                    f\"{len(missing)} of their files are absent (first {missing[0]!r})\")\n            return Path(root)\n    return None\n\n\n# How a member is read at inference. Overlapping windows over the slices the cache\n# already holds cost forward passes and no extra decoding, which is the cheap direction\n# to spend; and averaging probabilities rather than logits is an arithmetic mean of risk\n# rather than a geometric mean of odds, which orders studies differently. Both were\n# chosen by measuring them on the folds each member held out rather than by argument.\nTTA_OVERLAP = True\nTTA_POOL = \"prob\"\n\n# Public-frontier target pooling: the 0.899 notebook keeps the strongest window for\n# focal findings, and averages the two strongest windows for ACL/MCL. Its public\n# Pilkwang family is the same 20-member package used here, so these pooling decisions\n# transfer directly. Preserve the independently validated original-view Synovitis\n# route while retaining jitter for every other target.\nPUBLIC_FRONTIER_TARGET_POOL = {\n    \"Fracture\": \"max\",\n    \"Contusion\": \"max\",\n    \"Medial Meniscus\": \"max\",\n    \"Lateral Meniscus\": \"max\",\n    \"ACL\": \"top2\",\n    \"MCL\": \"top2\",\n    \"Baker's\": \"max\",\n}\nTTA_TARGET_POOL = {**PUBLIC_FRONTIER_TARGET_POOL, \"Synovitis\": \"original_mean\"}\n\n# V25 uses the 58 complete annotation rows from the official train.csv, joined by\n# StudyInstanceUID to the OOF predictions. The older y_derived artifact disagrees with\n# those expert labels in 145 cells and is not used for this decision. Repeated\n# target-stratified four-fold selection is macro-positive in 98% of 200 partitions and\n# isolates four targets; every other target returns to the native family. Lateral\n# Meniscus is shrunk from the selected median 1.00 to 0.75.\n# Convert desired final fraction f to each of four legacy-fold member weights:\n#     4*w/(20+4*w)=f -> w=5*f/(1-f).\nLEGACY_MEMBER_WEIGHT_BY_TARGET = {\n    \"Lateral Meniscus\": 15.0,           # final legacy fraction 0.75\n    \"Medial OA\": 2.5,                   # final legacy fraction 1/3\n    \"Lateral OA\": 15.0,                 # final legacy fraction 0.75\n    \"Contusion\": 5.0,                   # final legacy fraction 0.50\n}\n\n\ndef window_starts(n_slice, group, overlap=None):\n    \"\"\"Where each TTA window begins.\"\"\"\n    overlap = TTA_OVERLAP if overlap is None else overlap\n    if overlap and n_slice >= group:\n        return list(range(n_slice - group + 1))\n    return [g * group for g in range(max(n_slice // group, 1))]\n\n\ndef apply_target_window_pool(values, probs, logits, original_probs, mapping, target_idx):\n    \"\"\"Apply a target-specific pooling map to one batch in place.\"\"\"\n    for target, mode in mapping.items():\n        j = target_idx[target]\n        if mode == \"max\":\n            values[:, j] = probs[:, :, j].max(0).values\n        elif mode == \"mean\":\n            values[:, j] = probs[:, :, j].mean(0)\n        elif mode == \"logit_mean\":\n            values[:, j] = torch.sigmoid(logits[:, :, j].mean(0))\n        elif mode == \"original_mean\":\n            values[:, j] = original_probs[:, :, j].mean(0)\n        elif mode in (\"top2\", \"top3\"):\n            k = min(int(mode[3:]), probs.shape[0])\n            values[:, j] = probs[:, :, j].topk(k, dim=0).values.mean(0)\n        else:\n            raise ValueError(f\"unknown TTA pooling mode for {target}: {mode}\")\n    return values\n\n\n@torch.no_grad()\ndef predict_member(model, cache, mask, idx, dev, img_size, group=None, pool=None,\n                   starts=None, jitter=False, jitter_seed=SEED,\n                   return_public_frontier=False):\n    \"\"\"Predict one member; pool views within windows and windows across a study.\"\"\"\n    group = GROUP if group is None else group\n    pool = TTA_POOL if pool is None else pool\n    starts = window_starts(cache.shape[2], group) if starts is None else list(starts)\n    if not starts:\n        raise ValueError(\"predict_member was given no windows to average over\")\n    target_idx = {t: j for j, t in enumerate(TARGETS)}\n    unknown = (set(TTA_TARGET_POOL) | set(PUBLIC_FRONTIER_TARGET_POOL)) - set(target_idx)\n    if unknown:\n        raise ValueError(f\"unknown target(s) in TTA_TARGET_POOL: {unknown}\")\n\n    jitter_gen = torch.Generator(device=dev)\n    jitter_gen.manual_seed(int(jitter_seed) % (2**63 - 1))\n    model.eval()\n    out, public_frontier_out = [], []\n    for b in range(0, len(idx), EVAL_BATCH):\n        sel = idx[b:b + EVAL_BATCH]\n        m = torch.from_numpy(mask[sel]).to(dev)\n        win_probs, win_logits, win_original_probs = [], [], []\n        for st in starts:\n            rows = torch.from_numpy(\n                np.ascontiguousarray(cache[sel, :, st:st + group])).to(dev)\n            views = [rows] + ([augment(rows, generator=jitter_gen)] if jitter else [])\n            view_probs, view_logits = [], []\n            for view in views:\n                with torch.autocast(\"cuda\", enabled=dev.type == \"cuda\"):\n                    z = model(view, m, img_size).float()\n                view_logits.append(z)\n                view_probs.append(torch.sigmoid(z))\n            win_logits.append(torch.stack(view_logits).mean(0))\n            win_probs.append(torch.stack(view_probs).mean(0))\n            # views[0] is always the unaugmented acquisition. Preserve it so a\n            # target can opt out of jitter without another encoder forward pass.\n            win_original_probs.append(view_probs[0])\n\n        probs = torch.stack(win_probs)      # [window, batch, target]\n        logits = torch.stack(win_logits)\n        original_probs = torch.stack(win_original_probs)\n        v = (torch.sigmoid(logits.mean(0)) if pool == \"logit\" else probs.mean(0))\n        v = apply_target_window_pool(\n            v, probs, logits, original_probs, TTA_TARGET_POOL, target_idx\n        )\n        out.append(v.cpu().numpy())\n        if return_public_frontier:\n            public_v = apply_target_window_pool(\n                original_probs.mean(0), original_probs, logits, original_probs,\n                PUBLIC_FRONTIER_TARGET_POOL, target_idx,\n            )\n            public_frontier_out.append(public_v.cpu().numpy())\n    primary = (np.concatenate(out) if out else\n               np.zeros((0, len(TARGETS)), np.float32))\n    if not return_public_frontier:\n        return primary\n    public_frontier = (np.concatenate(public_frontier_out) if public_frontier_out else\n                       np.zeros((0, len(TARGETS)), np.float32))\n    return primary, public_frontier\n\n\n# HF from_pretrained mutates process-global state and is not guaranteed thread-safe, so\n# model construction, weight loading and the fingerprint check are serialised; only\n# inference -- the expensive part -- runs on both devices at once.\nBUILD_LOCK = threading.Lock()\nSTATE_LOCK = threading.Lock()\n\n# An independently trained four-fold bundle joins the vote at reduced weight: a second\n# training run (public 0.836 on its own) adds decorrelated errors, which is the only thing an\n# inference-only run can add that the 20-member package does not already have. 0.5 per\n# fold puts the bundle at ~11% of the total vote -- roughly its quality gap.\nLEGACY_BUNDLE_FILE = \"rsna_20260807_v1.pt\"\nLEGACY_WEIGHT = 0.5\n\n\ndef find_legacy_bundle():\n    base = Path(\"/kaggle/input\")\n    if not base.is_dir():\n        return None\n    for root, dirs, files in os.walk(base):\n        dirs[:] = [d for d in dirs if d not in (\"train_series\", \"test_series\")]\n        if LEGACY_BUNDLE_FILE in files:\n            return Path(root) / LEGACY_BUNDLE_FILE\n    return None\n\n\ndef legacy_group_members():\n    \"\"\"The four-fold bundle as extra, lower-weight members under RULES_LEGACY.\n\n    The package's legacy pixel rules exist precisely to reproduce what this bundle was\n    fitted on (dominant-axis slice order, corner-x laterality at 5 mm, T1 slot fallback,\n    zero decode fill, 160 mm crop, central 60% band). The bundle predates fingerprints,\n    which is accepted loudly and priced into its reduced weight; a fold whose state dict\n    does not load, or whose predictions are degenerate, is dropped and costs its own\n    vote only.\n    \"\"\"\n    p = find_legacy_bundle()\n    if p is None:\n        log(\"no legacy bundle attached; blending skipped\")\n        return {}\n    try:\n        b = torch.load(p, map_location=\"cpu\", weights_only=False)\n        folds = b.get(\"fold_states\") or []\n        b_slots = [tuple(s)[0] for s in b.get(\"slots\", SLOTS)]\n        if list(b.get(\"targets\", TARGETS)) != TARGETS or b_slots != [s[0] for s in SLOTS]:\n            log(f\"legacy bundle {p.name}: target/slot contract differs; blending skipped\")\n            return {}\n        gr, n_gr = int(b.get(\"group\", 3)), int(b.get(\"n_group\", 3))\n        variant = str(b.get(\"model_variant\", \"dinov2-small\")).split(\"-\")[-1]\n        key = json.dumps({\"img\": int(b.get(\"img\", 224)), \"group\": gr,\n                          \"slices\": gr * n_gr, \"crop_mm\": 160.0, \"band\": [0.20, 0.80],\n                          \"rules\": RULES_LEGACY, \"slots\": [s[0] for s in SLOTS]},\n                         sort_keys=True)\n        ms = [{\"id\": f\"legacy-f{f.get('fold', k)}\", \"fold\": f.get(\"fold\", k),\n               \"state\": f[\"state_dict\"], \"holdout\": None, \"weight\": LEGACY_WEIGHT,\n               \"target_weight\": [LEGACY_MEMBER_WEIGHT_BY_TARGET.get(t, 0.0)\n                                 for t in TARGETS],\n               \"pixel_group\": key,\n               \"config\": {\"unfreeze_last\": 6,\n                          \"variant\": \"base\" if variant == \"base\" else \"small\",\n                          \"pool\": \"cls_mean_focal\", \"prior\": True}}\n              for k, f in enumerate(folds)]\n        if ms:\n            active = sorted(set(LEGACY_MEMBER_WEIGHT_BY_TARGET.values()))\n            log(f\"legacy bundle {p.name}: {len(ms)} fold(s) join with \"\n                f\"target-specific per-member weights {active}\")\n        return {key: ms} if ms else {}\n    except Exception as exc:\n        log(f\"legacy bundle unusable ({type(exc).__name__}: {exc}); blending skipped\")\n        return {}\n\n\ndef _run_member(path, m, dev, Cte, Mte, idx, starts, jitter):\n    \"\"\"Load, verify and predict one member on one device. Returns (pred, timings).\"\"\"\n    t0 = time.time()\n    with BUILD_LOCK:\n        if \"state\" in m:\n            state, fp = m[\"state\"], None\n        else:\n            ck = torch.load(Path(path) / m[\"file\"], map_location=\"cpu\",\n                            weights_only=False)\n            state, fp = ck[\"model\"], ck.get(\"fingerprint\")\n        model = build_model(int(m[\"config\"][\"unfreeze_last\"]),\n                            variant=m[\"config\"][\"variant\"],\n                            pool=m[\"config\"].get(\"pool\", \"cls_mean\"),\n                            prior=bool(m[\"config\"].get(\"prior\", False))).to(dev)\n        model.load_state_dict(state)\n        if fp is not None:\n            check_fingerprint(model, dev, IMG, fp, tag=f\"{m['id']}: \")\n        else:\n            log(f\"  {m['id']}: no stored fingerprint (legacy bundle) -- \"\n                f\"accepted at reduced weight\")\n    t_ready = time.time()\n    jitter_seed = SEED + int(hashlib.sha256(str(m[\"id\"]).encode()).hexdigest()[:8], 16)\n    public_member = \"state\" not in m\n    predicted = predict_member(\n        model, Cte, Mte, idx, dev, IMG, starts=starts, jitter=jitter,\n        jitter_seed=jitter_seed, return_public_frontier=public_member,\n    )\n    if public_member:\n        p, public_p = predicted\n    else:\n        p, public_p = predicted, None\n    t_done = time.time()\n    del model, state\n    gc.collect()\n    if dev.type == \"cuda\":\n        with torch.cuda.device(dev):\n            torch.cuda.empty_cache()\n    passes = len(starts) * (2 if jitter else 1)\n    return p, public_p, (t_ready - t0, (t_done - t_ready) / max(passes, 1))\n\n\ndef _combine(per_member):\n    \"\"\"Target-wise weighted mean of per-member percentile ranks.\"\"\"\n    all_ids = sorted({s for m in per_member for s in m[\"ids\"]})\n    pos = {s: i for i, s in enumerate(all_ids)}\n    acc = np.zeros((len(all_ids), len(TARGETS)), np.float64)\n    tot = np.zeros(len(TARGETS), np.float64)\n    for m in per_member:\n        target_weight = m.get(\"target_weight\")\n        w = np.asarray(target_weight if target_weight is not None else\n                       [float(m.get(\"weight\", 1.0))] * len(TARGETS),\n                       dtype=np.float64)\n        if w.shape != (len(TARGETS),) or np.any(w < 0):\n            raise ValueError(f\"invalid target weights for {m.get('id')}: {w}\")\n        r = pd.DataFrame(m[\"pred\"]).rank(pct=True).to_numpy()\n        acc[[pos[s] for s in m[\"ids\"]]] += r * w[None, :]\n        tot += w\n    if np.any(tot <= 0):\n        raise ValueError(f\"at least one target has no ensemble vote: {tot}\")\n    return all_ids, acc / tot[None, :]\n\n\ndef infer_from_package(path, dev=None):\n    \"\"\"Predict the test split from an attached package of trained members.\n\n    Identical pixel contract to the reference implementation -- same caches, same\n    windows, same fingerprints, same rank transform -- with executive changes only:\n\n    1. A work queue over the devices: each GPU pops the next member when free (the\n       reference ran one device and surrendered TTA windows, then members: 0.847).\n    2. submission.csv is rewritten after every banked member, so a run killed at any\n       point still submits the best partial ensemble instead of the 0.5 benchmark.\n    3. A member that fails on one device is retried once on the other, then dropped;\n       a dropped member costs one vote, never the run.\n    4. An independently trained legacy bundle joins as four target-selective reduced-weight members.\n    5. When the time estimate says the whole remaining ensemble fits with room to\n       spare, each window gains one jittered TTA view (same family as training aug).\n    \"\"\"\n    man = json.loads((Path(path) / \"manifest.json\").read_text())\n    members = man[\"members\"]\n    log(f\"weights package: {len(members)} member(s) from {path}; \"\n        f\"{len(DEVS)} device(s)\")\n\n    test_df = pd.read_csv(ROOT / \"test.csv\")\n    test_series = pd.read_csv(ROOT / \"test_series.csv\")\n    plane_map = dict(zip(test_series[\"SeriesInstanceUID\"],\n                         test_series[\"Anatomical_Plane\"]))\n    hte = annotate(walk(\"test_series\"))\n    log(f\"test header pass: {len(hte)} series\")\n\n    groups = {}\n    for m in members:\n        groups.setdefault(m[\"pixel_group\"], []).append(m)\n    # Legacy votes come last: if time binds after all, the weakest votes are the ones\n    # surrendered, not the package's.\n    groups.update(legacy_group_members())\n\n    per_member, public_frontier_members = [], []\n    est = {\"fixed\": None, \"win\": None}\n\n    def bank(m, ids, pred, starts, jitter, public_pred=None):\n        if float(np.std(pred)) < 1e-9:\n            log(f\"  {m['id']}: degenerate predictions; not banked\")\n            return\n        with STATE_LOCK:\n            per_member.append({\"id\": m[\"id\"], \"ids\": ids, \"pred\": pred,\n                               \"weight\": m.get(\"weight\", 1.0),\n                               \"target_weight\": m.get(\"target_weight\"),\n                               \"holdout\": m.get(\"holdout\")})\n            if public_pred is not None and len(starts) == len(starts_full):\n                if float(np.std(public_pred)) < 1e-9:\n                    raise WeightsError(f\"{m['id']}: degenerate public-frontier prediction\")\n                public_frontier_members.append(\n                    {\"id\": m[\"id\"], \"ids\": ids, \"pred\": public_pred}\n                )\n            elif public_pred is not None:\n                log(\n                    f\"  {m['id']}: public-frontier vote omitted because only \"\n                    f\"{len(starts)} / {len(starts_full)} windows completed\"\n                )\n            all_ids, acc = _combine(per_member)\n            write_submission(acc, all_ids, test_df, \"submission.csv\")\n            log(f\"  banked {m['id']} fold {m.get('fold', '?')} \"\n                f\"({len(starts)} window(s){', jitter' if jitter else ''}); \"\n                f\"submission.csv = weighted rank mean of {len(per_member)} member(s)\")\n\n    for gi, (key, gm) in enumerate(groups.items(), 1):\n        cfg = json.loads(key)\n        adopt_config_globals(cfg)\n        log(f\"decode group {gi}/{len(groups)}: {cfg['img']}px x {cfg['slices']} slices, \"\n            f\"crop {cfg['crop_mm']} mm -> {len(gm)} member(s)\")\n        st_te, Cte, Mte = build_cache(pick_slots(hte, plane_map), plane_map,\n                                      lat_of(hte, \"test \"), f\"test g{gi}\")\n        idx = np.arange(len(st_te))\n\n        starts_full = window_starts(Cte.shape[2], GROUP)\n        pending = sorted(gm, key=lambda m: -(m.get(\"holdout\") or 0))\n        left_after = sum(len(g) for j, (_, g) in enumerate(groups.items(), 1) if j > gi)\n\n        def pop_next():\n            \"\"\"Next member, its window count, and whether jitter TTA is affordable.\n\n            Windows are surrendered before members; jitter is granted only when the\n            estimate says the whole remaining ensemble fits at double passes inside\n            60% of the room. The question asked is whether ONE more member fits.\n            \"\"\"\n            with STATE_LOCK:\n                if not pending:\n                    return None, None, False\n                left = TIME_BUDGET - (time.time() - T0)\n                remaining = len(pending) + left_after\n                slots_left = -(-remaining // len(DEVS))       # ceil: concurrent slots\n                starts, jit = starts_full, False\n                if est[\"fixed\"] is not None and est[\"win\"] is not None:\n                    afford = max(left * 0.9, 0.0)\n                    room = afford / max(slots_left, 1)\n                    if est[\"fixed\"] + est[\"win\"] > room:\n                        log(f\"  {left / 60:.0f} min left: surrendering \"\n                            f\"{len(pending)} member(s); not one more fits\")\n                        pending.clear()\n                        return None, None, False\n                    jit = (est[\"fixed\"] + 2 * len(starts_full) * est[\"win\"]\n                           <= room * 0.6)\n                    per_win = est[\"win\"] * (2 if jit else 1)\n                    n_win = (int((room - est[\"fixed\"]) / per_win)\n                             if per_win > 0 else len(starts_full))\n                    n_win = max(1, min(len(starts_full), n_win))\n                    if n_win < len(starts_full):\n                        mid = (len(starts_full) - n_win) // 2\n                        starts = starts_full[mid:mid + n_win]\n                return pending.pop(0), starts, jit\n\n        def worker(dev):\n            others = [d for d in DEVS if d is not dev]\n            while True:\n                m, starts, jit = pop_next()\n                if m is None:\n                    return\n                for attempt, d in enumerate([dev] + others[:1]):\n                    try:\n                        p, public_p, (fs, ws) = _run_member(\n                            path, m, d, Cte, Mte, idx, starts, jit\n                        )\n                        with STATE_LOCK:\n                            est[\"fixed\"], est[\"win\"] = fs, ws\n                        bank(m, st_te, p, starts, jit, public_p)\n                        break\n                    except Exception as exc:\n                        log(f\"  MEMBER {m['id']} failed on {d} \"\n                            f\"({type(exc).__name__}: {exc}); \"\n                            + (\"retrying on peer device\" if attempt == 0 and others\n                               else \"dropped -- costs one vote, not the run\"))\n                        if d.type == \"cuda\":\n                            with torch.cuda.device(d):\n                                torch.cuda.empty_cache()\n\n        threads = [threading.Thread(target=worker, args=(d,)) for d in DEVS]\n        for t in threads:\n            t.start()\n        for t in threads:\n            t.join()\n\n        del Cte, Mte\n        gc.collect()\n\n    if not per_member:\n        raise WeightsError(\"no member produced predictions; submission stays at 0.5\")\n\n    all_ids, acc = _combine(per_member)\n    sub = write_submission(acc, all_ids, test_df, \"submission.csv\")\n    log(f\"final submission.csv = weighted rank mean of {len(per_member)} member(s); \"\n        f\"{sub.shape}; nulls {int(sub[TARGETS].isna().sum().sum())}\")\n    if len(public_frontier_members) == len(members):\n        frontier_ids, frontier_acc = _combine(public_frontier_members)\n        frontier_sub = write_submission(\n            frontier_acc, frontier_ids, test_df, \"submission_public_0899.csv\"\n        )\n        log(\n            f\"submission_public_0899.csv = exact no-jitter public-frontier rank mean \"\n            f\"of {len(public_frontier_members)} member(s); {frontier_sub.shape}; \"\n            f\"nulls {int(frontier_sub[TARGETS].isna().sum().sum())}\"\n        )\n    else:\n        log(\n            f\"public-frontier fallback not emitted: {len(public_frontier_members)} / \"\n            f\"{len(members)} required public members completed\"\n        )\n    return sub\n\n\ndef adopt_config_globals(cfg):\n    \"\"\"Point the pixel path at what one group of members was fitted on.\"\"\"\n    global IMG, CACHE_IMG, GROUP, CACHE_SLICES, N_GROUP, CROP_MM, SLICE_BAND, RULES\n    CACHE_IMG = IMG = int(cfg[\"img\"])\n    GROUP = int(cfg[\"group\"])\n    CACHE_SLICES = int(cfg[\"slices\"])\n    N_GROUP = max(CACHE_SLICES // GROUP, 1)\n    CROP_MM = float(cfg[\"crop_mm\"])\n    SLICE_BAND = tuple(float(x) for x in cfg[\"band\"])\n    # The four decisions that change what a slice is. A member fitted under one reading\n    # and decoded under another gets pixels its weights never saw, with every shape\n    # still agreeing, so an unrecognised name is refused rather than defaulted.\n    rules = cfg.get(\"rules\") or RULES_NATIVE\n    unknown = {k: v for k, v in rules.items()\n               if k not in RULES_NATIVE\n               or v not in (RULES_NATIVE[k], RULES_LEGACY[k])}\n    if unknown:\n        raise WeightsError(f\"the members record pixel rules this pipeline cannot \"\n                           f\"reproduce: {unknown}\")\n    RULES = {**RULES_NATIVE, **rules}\n    if [s[0] for s in SLOTS] != list(cfg[\"slots\"]):\n        raise WeightsError(\n            f\"the members were fitted on slots {cfg['slots']} and this pipeline defines \"\n            f\"{[s[0] for s in SLOTS]}; a weight would be read against the wrong slot\")\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-10T06:00:16.532331Z","iopub.status.busy":"2026-08-10T06:00:16.532034Z","iopub.status.idle":"2026-08-10T06:00:16.589416Z","shell.execute_reply":"2026-08-10T06:00:16.588615Z"},"papermill":{"duration":0.080599,"end_time":"2026-08-10T06:00:16.591022+00:00","exception":false,"start_time":"2026-08-10T06:00:16.510423+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"a01d8443","cell_type":"code","source":"def take_group(cache_rows, g):\n    \"\"\"Slice GROUP consecutive channels out of the cached slices.\"\"\"\n    return cache_rows[:, :, g * GROUP:(g + 1) * GROUP]\n\n\ndef augment(imgs, generator=None):\n    \"\"\"A small rigid jitter and an intensity scale, applied to a whole bag at once.\n\n    Neither flip is available here, and for different reasons. A horizontal flip would\n    reintroduce the nuisance axis that the laterality normalisation removed - it would\n    undo, once per batch, what the header pass was run to establish.\n\n    A vertical flip is not a nuisance axis at all. A knee is acquired in a canonical\n    orientation, and no study in this corpus looks like its own vertical mirror. An\n    augmentation is meant to cover directions along which the label does not change; this\n    one moves the input off the distribution the encoder will be asked about, which is a\n    different thing. Where a finding sits in the frame is also information rather than\n    noise - a Baker cyst is identified by lying in the popliteal fossa, not by its\n    appearance alone.\n\n    What is left is jitter that no label depends on: a few degrees of rotation, a few\n    per cent of scale and translation. That still prevents memorising the exact framing,\n    which is what an augmentation is for, while leaving the anatomy where it was.\n    \"\"\"\n    # A bag arrives as [study, slot, GROUP, IMG, IMG]: five axes, not four. The warp is\n    # a 2-D operation, so the two leading axes are folded together and restored after -\n    # every slot image is an independent acquisition and gets its own jitter.\n    lead = imgs.shape[:-3]\n    x = imgs.reshape(-1, *imgs.shape[-3:]).float()\n    n, dev = x.shape[0], x.device\n\n    rot = (torch.rand(n, device=dev, generator=generator) - 0.5) * 2 * (AUG_ROT_DEG * np.pi / 180)\n    # Zoom in only. `border` padding repeats the edge row outward, and the edge of this\n    # crop is where the popliteal fossa sits; zooming out would fabricate tissue exactly\n    # where a Baker cyst is looked for.\n    sc = 1.0 + torch.rand(n, device=dev, generator=generator) * AUG_SCALE\n    tx = (torch.rand(n, device=dev, generator=generator) - 0.5) * 2 * AUG_SHIFT\n    ty = (torch.rand(n, device=dev, generator=generator) - 0.5) * 2 * AUG_SHIFT\n    cos, sin = torch.cos(rot) / sc, torch.sin(rot) / sc\n    theta = torch.zeros(n, 2, 3, device=dev, dtype=torch.float32)\n    theta[:, 0, 0], theta[:, 0, 1], theta[:, 0, 2] = cos, -sin, tx\n    theta[:, 1, 0], theta[:, 1, 1], theta[:, 1, 2] = sin, cos, ty\n    grid = F.affine_grid(theta, x.shape, align_corners=False)\n    x = F.grid_sample(x, grid, mode=\"bilinear\", padding_mode=\"border\", align_corners=False)\n\n    scale = 1.0 + (torch.rand(n, 1, 1, 1, device=dev, generator=generator) - 0.5) * 2 * AUG_INTENSITY\n    x = (x * scale).clamp(0, 255)\n    return x.reshape(*lead, *x.shape[-3:]).to(imgs.dtype)\n\n\n@torch.no_grad()\ndef predict(model, cache, mask, idx, dev, img_size=None):\n    \"\"\"Average the logits over the groups of each slot.\n\n    Training sees one group at a time, which acts as augmentation along the stack;\n    inference averages over all of them, so the prediction does not depend on which\n    group a single draw happened to pick. Where the cache holds one group per slot the\n    two coincide.\n    \"\"\"\n    model.eval()\n    out = []\n    for b in range(0, len(idx), EVAL_BATCH):\n        sel = idx[b:b + EVAL_BATCH]\n        m = torch.from_numpy(mask[sel]).to(dev)\n        acc = None\n        for g in range(N_GROUP):\n            # Gathered a group at a time rather than whole and then sliced. The two are\n            # the same pixels, but taking the whole of a study out of the cache allocates\n            # every slice it holds - most of which this pass will not look at until a\n            # later iteration, by which time they have been fetched again. Measured over\n            # a cache of twelve slices, the difference between the two is the difference\n            # between the step being bound by memory and being bound by the encoder.\n            rows = torch.from_numpy(np.ascontiguousarray(\n                cache[sel, :, g * GROUP:(g + 1) * GROUP])).to(dev)\n            with torch.autocast(\"cuda\", enabled=dev.type == \"cuda\"):\n                z = model(rows, m, img_size).float()\n            acc = z if acc is None else acc + z\n        out.append(torch.sigmoid(acc / N_GROUP).cpu().numpy())\n    return np.concatenate(out) if out else np.zeros((0, len(TARGETS)), np.float32)\n\n\ndef macro_auc(y, p):\n    from sklearn.metrics import roc_auc_score\n    return float(np.nanmean([roc_auc_score(y[:, j], p[:, j])\n                             if len(set(y[:, j])) > 1 else np.nan\n                             for j in range(y.shape[1])]))\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-10T06:00:16.632391Z","iopub.status.busy":"2026-08-10T06:00:16.631813Z","iopub.status.idle":"2026-08-10T06:00:16.643938Z","shell.execute_reply":"2026-08-10T06:00:16.643393Z"},"papermill":{"duration":0.034082,"end_time":"2026-08-10T06:00:16.645323+00:00","exception":false,"start_time":"2026-08-10T06:00:16.611241+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"28fcd4ab","cell_type":"code","source":"def write_submission(pred, studies, test_df, path):\n    \"\"\"Write one submission file from a prediction matrix.\n\n    Predictions are converted to per-column ranks first: the metric reads only order, so\n    ranks discard nothing, and they make files from different configurations directly\n    comparable and safe to average.\n    \"\"\"\n    sub = pd.DataFrame(pd.DataFrame(pred).rank(pct=True).values, columns=TARGETS)\n    sub.insert(0, \"StudyInstanceUID\", studies)\n    sub = test_df[[\"StudyInstanceUID\"]].merge(sub, on=\"StudyInstanceUID\", how=\"left\")\n    sub[TARGETS] = sub[TARGETS].fillna(0.5)\n    sub.to_csv(path, index=False)\n    return sub\n\n\ndef write_benchmark_submission():\n    \"\"\"Write the 0.5 benchmark file immediately.\n\n    A submission that never writes scores nothing at all, which is strictly worse than\n    scoring badly. The try/except around main() covers exceptions, but a kill for memory\n    is a SIGKILL and never reaches it. So a valid file exists from the first second and\n    is overwritten only once real predictions are ready.\n    \"\"\"\n    t = pd.read_csv(ROOT / \"test.csv\")\n    for c in TARGETS:\n        t[c] = 0.5\n    t.to_csv(\"submission.csv\", index=False)\n\n\ndef _v37_validate_submission(path, test_df, tag):\n    \"\"\"Read one attached prediction file only after its full contract passes.\"\"\"\n    path = Path(path)\n    frame = pd.read_csv(path)\n    expected = [\"StudyInstanceUID\"] + TARGETS\n    if list(frame.columns) != expected:\n        raise ValueError(f\"{tag}: columns differ from the competition contract\")\n    if len(frame) != len(test_df) or not frame[\"StudyInstanceUID\"].is_unique:\n        raise ValueError(f\"{tag}: row count or StudyInstanceUID uniqueness failed\")\n    if set(frame[\"StudyInstanceUID\"].astype(str)) != set(test_df[\"StudyInstanceUID\"].astype(str)):\n        raise ValueError(f\"{tag}: StudyInstanceUID set differs from test.csv\")\n    values = frame[TARGETS].to_numpy(np.float64)\n    if not np.isfinite(values).all():\n        raise ValueError(f\"{tag}: non-finite prediction\")\n    return test_df[[\"StudyInstanceUID\"]].merge(frame, on=\"StudyInstanceUID\", how=\"left\")\n\n\ndef _v37_find_yash_submission():\n    \"\"\"Find the output mounted from the exact public Yash notebook source.\"\"\"\n    candidates = []\n    local = globals().get(\"YASH_LOCAL_SOURCE_DIR\")\n    if local:\n        candidates.append(Path(local) / \"submission.csv\")\n    root = Path(\"/kaggle/input\")\n    candidates.append(root / \"rsna-knee-infer-v1\" / \"submission.csv\")\n    if root.is_dir():\n        candidates.extend(meta.parent / \"submission.csv\"\n                          for meta in root.glob(\"**/infer_meta.json\"))\n    seen = set()\n    for path in candidates:\n        key = str(path)\n        if key in seen or not path.is_file():\n            continue\n        seen.add(key)\n        meta_path = path.with_name(\"infer_meta.json\")\n        if meta_path.is_file():\n            meta = json.loads(meta_path.read_text())\n            if int(meta.get(\"errors\", -1)) != 0:\n                raise ValueError(f\"Yash source reports {meta.get('errors')} inference errors\")\n        return path\n    raise FileNotFoundError(\"the attached yashbishnoi98/rsna-knee-infer-v1 output is absent\")\n\n\ndef run_yash_public_ensemble():\n    \"\"\"Bank the public image specialist and a conservative rank ensemble.\n\n    The Yash source is an independently trained EfficientNet-B3, five-fold,\n    plane-aware image family. The local public fallback is the independently\n    published twenty-member DINO family. A 0.55/0.45 Borda blend keeps Yash as\n    the stronger voter but lets concordant DINO evidence resolve close orderings.\n    The exact public source and the pre-blend native primary are both retained.\n    \"\"\"\n    import shutil\n\n    test_df = pd.read_csv(ROOT / \"test.csv\")\n    native_path = Path(\"submission.csv\")\n    public_path = Path(\"submission_public_0899.csv\")\n    native = _v37_validate_submission(native_path, test_df, \"native V36\")\n    public = _v37_validate_submission(public_path, test_df, \"public DINO family\")\n    yash_path = _v37_find_yash_submission()\n    yash = _v37_validate_submission(yash_path, test_df, \"Yash public image family\")\n    meta_path = yash_path.with_name(\"infer_meta.json\")\n    if meta_path.is_file():\n        meta = json.loads(meta_path.read_text())\n        if int(meta.get(\"studies\", -1)) != len(test_df):\n            raise ValueError(\"Yash source study count differs from test.csv\")\n\n    shutil.copyfile(native_path, \"submission_native_v36.csv\")\n    shutil.copyfile(yash_path, \"submission_yash_reference.csv\")\n    yr = yash[TARGETS].rank(pct=True).to_numpy(np.float64)\n    dr = public[TARGETS].rank(pct=True).to_numpy(np.float64)\n    blend = 0.55 * yr + 0.45 * dr\n    result = test_df[[\"StudyInstanceUID\"]].copy()\n    result[TARGETS] = blend\n    if result.shape != yash.shape or not np.isfinite(result[TARGETS].to_numpy()).all():\n        raise AssertionError(\"invalid Yash/DINO rank blend\")\n    candidate_path = Path(\"submission_yash_dino_rankblend.csv\")\n    result.to_csv(candidate_path, index=False)\n    reread = _v37_validate_submission(candidate_path, test_df, \"Yash/DINO rank blend\")\n    changed = sum(\n        tuple(reread[target].rank(method=\"first\")) !=\n        tuple(yash[target].rank(method=\"first\"))\n        for target in TARGETS\n    )\n    if changed == 0:\n        raise AssertionError(\"Yash/DINO blend is rank-identical to its Yash parent\")\n    temp_path = Path(\"submission_v37_yash_dino.tmp.csv\")\n    reread.to_csv(temp_path, index=False)\n    temp_path.replace(native_path)\n    log(f\"Yash public family banked; V37 primary = 0.55 Yash / 0.45 public DINO \"\n        f\"rank blend ({changed} target orderings differ from Yash); exact Yash and \"\n        \"native V36 outputs retained\")\n    return True\n\n\ndef main():\n    write_benchmark_submission()\n\n    # Weights, if any were attached; otherwise the run learns its own below. Both paths\n    # are kept because the second is what makes this notebook readable on its own - a\n    # fork with nothing attached still trains and still scores - and because the first\n    # cannot be checked by anyone who does not have the package.\n    pkg = find_weights()\n    if pkg is not None:\n        dev = DEVS[0]\n        infer_from_package(pkg, dev)\n\n        # The exact no-jitter target-pooling recipe independently completed at 0.899 in\n        # the public frontier.  Earlier versions retained it as a secondary artifact and\n        # then promoted less certain specialists.  Make the evidence-backed arm primary\n        # only after the complete hidden UID/schema/finite-value contract passes.\n        try:\n            test_df = pd.read_csv(ROOT / \"test.csv\")\n            native_path = Path(\"submission.csv\")\n            public_path = Path(\"submission_public_0899.csv\")\n            native = _v37_validate_submission(native_path, test_df, \"native 24-member\")\n            public = _v37_validate_submission(public_path, test_df, \"public DINO frontier\")\n            native.to_csv(\"submission_native_v38.csv\", index=False)\n            public.to_csv(native_path, index=False)\n            promoted = _v37_validate_submission(native_path, test_df, \"V40 primary\")\n            if not promoted.equals(public):\n                raise AssertionError(\"V40 serialization differs from validated public frontier\")\n            log(\"V40 primary = exact no-jitter public-frontier target pooling; \"\n                \"native 24-member output retained\")\n        except Exception as public_frontier_error:\n            log(f\"public-frontier promotion skipped safely: {public_frontier_error}\")\n            traceback.print_exc()\n        log(\"done\")\n        return\n\n    # Settle where the labels come from before anything expensive runs. The check costs\n    # one CSV header read; discovering the same problem after the cache is built would\n    # cost the whole decode pass, and discovering it never would cost the run.\n    read_labels(pd.read_csv(ROOT / \"train.csv\", usecols=[\"StudyInstanceUID\", \"Report\"]))\n\n    test_df = pd.read_csv(ROOT / \"test.csv\")\n    test_series = pd.read_csv(ROOT / \"test_series.csv\")\n    train_df = pd.read_csv(ROOT / \"train.csv\")\n    train_series = pd.read_csv(ROOT / \"train_series.csv\")\n    log(f\"train {train_df.shape} test {test_df.shape}\")\n\n    both = pd.concat([train_series, test_series])\n    plane_map = dict(zip(both[\"SeriesInstanceUID\"], both[\"Anatomical_Plane\"]))\n\n    log(\"header pass: test\")\n    hte = annotate(walk(\"test_series\"))\n    log(f\"  {len(hte)} test series\")\n    log(\"header pass: train\")\n    htr = annotate(walk(\"train_series\"))\n    log(f\"  {len(htr)} train series\")\n\n    slots_te, slots_tr = pick_slots(hte, plane_map), pick_slots(htr, plane_map)\n    cov = pd.Series([len(v) for v in slots_tr.values()]).describe()\n    log(f\"train slots per study: mean {cov['mean']:.2f} min {cov['min']:.0f} \"\n        f\"max {cov['max']:.0f}\")\n\n    st_tr, Ctr, Mtr = build_cache(slots_tr, plane_map, lat_of(htr, \"train \"), \"train\")\n    st_te, Cte, Mte = build_cache(slots_te, plane_map, lat_of(hte, \"test \"), \"test\")\n\n    # ---- targets ---------------------------------------------------------- #\n    t_lab = time.time()\n    lab = read_labels(train_df)\n    log(f\"derived labels for {len(lab)} studies in {time.time() - t_lab:.1f}s\")\n\n    gold = train_df.set_index(\"StudyInstanceUID\")[TARGETS]\n    gold = gold[gold.notna().all(axis=1)]\n\n    Y = np.zeros((len(st_tr), len(TARGETS)), np.float32)\n    W = np.zeros_like(Y)\n    for i, st in enumerate(st_tr):\n        if st in gold.index:\n            Y[i], W[i] = gold.loc[st].values, 3.0\n        elif st in lab.index:\n            r = lab.loc[st]\n            Y[i] = r[TARGETS].values\n            W[i] = 0.25 + 0.75 * r[[t + \"__conf\" for t in TARGETS]].values\n    keep = np.where(W.sum(1) > 0)[0]\n    log(f\"supervised {len(keep)} of {len(st_tr)} studies (annotated {len(gold)})\")\n\n    # Grouped on report text: some reports are byte-identical across studies and yield\n    # one target vector for all of them, so splitting such a group scores the model on a\n    # target whose source it has already trained on.\n    import hashlib\n    rep = train_df.set_index(\"StudyInstanceUID\")[\"Report\"].fillna(\"\")\n    grp = np.array([int(hashlib.md5(rep.get(s, s).encode()).hexdigest()[:8], 16) % 5\n                    for s in st_tr])\n    va = np.array([i for i in keep if grp[i] == 0])\n    tr = np.array([i for i in keep if grp[i] != 0])\n    if len(va) == 0 or len(tr) < BATCH_STUDIES:\n        cut = max(1, len(keep) // 5)\n        va, tr = keep[:cut], keep[cut:]\n    log(f\"train {len(tr)} / holdout {len(va)} studies\")\n\n    # The annotated studies stay in training - they are the highest-quality labels in\n    # the corpus and there are too few to discard - so the honest annotation check uses\n    # only the ones that fell in the holdout. Evaluating on the rest would be scoring the\n    # model against examples it was trained on, at triple weight, with the true answer.\n    gpos = {s: i for i, s in enumerate(st_tr)}\n    va_set = set(va.tolist())\n    gi = np.array([gpos[s] for s in gold.index if s in gpos and gpos[s] in va_set])\n    gold_y = gold.loc[[st_tr[i] for i in gi]].values.astype(int) if len(gi) else None\n    yv = (Y[va] > 0.5).astype(int)\n    log(f\"annotation check: {len(gi)} of {len(gold)} annotated studies are in the holdout\")\n\n    # ---- fine-tune -------------------------------------------------------- #\n    dev = DEVS[0]\n    results, test_preds = {}, {}\n\n    for cfg in RUNS:\n        pitch = CROP_MM / cfg[\"img\"]\n        log(f\"=== {cfg['name']}: {cfg['img']} px, {pitch:.3f} mm/pixel, \"\n            f\"{pitch * 14:.2f} mm per patch token ===\")\n        torch.manual_seed(SEED)\n        model = build_model(UNFREEZE_LAST).to(dev)\n        opt = torch.optim.AdamW([\n            {\"params\": [p for p in model.backbone.parameters() if p.requires_grad],\n             \"lr\": LR_BACKBONE},\n            {\"params\": model.head.parameters(), \"lr\": LR_HEAD},\n        ], weight_decay=WEIGHT_DECAY)\n        steps = max(EPOCHS * (len(tr) // BATCH_STUDIES), 1)\n        sched = torch.optim.lr_scheduler.OneCycleLR(\n            opt, max_lr=[LR_BACKBONE, LR_HEAD], total_steps=steps, pct_start=0.15)\n        scaler = torch.amp.GradScaler(\"cuda\", enabled=dev.type == \"cuda\")\n\n        best, best_state, best_annot = -1.0, None, float(\"nan\")\n        for ep in range(EPOCHS):\n            model.train()\n            perm = np.random.permutation(tr)\n            tot, nstep = 0.0, 0\n            for b in range(0, len(perm) - BATCH_STUDIES + 1, BATCH_STUDIES):\n                sel = perm[b:b + BATCH_STUDIES]\n                rows = torch.from_numpy(Ctr[sel]).to(dev)\n                g = int(torch.randint(N_GROUP, (1,)).item())\n                imgs = augment(take_group(rows, g))\n                m = torch.from_numpy(Mtr[sel]).to(dev)\n                y = torch.from_numpy(Y[sel]).to(dev)\n                w = torch.from_numpy(W[sel]).to(dev)\n                with torch.autocast(\"cuda\", enabled=dev.type == \"cuda\"):\n                    loss = (F.binary_cross_entropy_with_logits(\n                        model(imgs, m, cfg[\"img\"]), y, reduction=\"none\") * w).mean()\n                opt.zero_grad(set_to_none=True)\n                scaler.scale(loss).backward()\n                scaler.step(opt)\n                scaler.update()\n                sched.step()\n                tot += loss.item()\n                nstep += 1\n\n            pv = predict(model, Ctr, Mtr, va, dev, cfg[\"img\"])\n            d = macro_auc(yv, pv)\n            g_auc = float(\"nan\")\n            if gold_y is not None and len(gi):\n                g_auc = macro_auc(gold_y, predict(model, Ctr, Mtr, gi, dev, cfg[\"img\"]))\n            log(f\"  epoch {ep + 1}/{EPOCHS}  loss {tot / max(nstep, 1):.4f}\"\n                f\"  holdout {d:.4f}  annot(n={len(gi)}) {g_auc:.4f}\")\n\n            # Selection reads the holdout alone. The annotation check is reported because\n            # it measures something different - agreement with a reading of the images\n            # rather than of the reports - but only a handful of annotated studies land\n            # in any one holdout, so its sampling error dwarfs the differences between\n            # epochs and it cannot arbitrate between them.\n            if d > best:\n                best, best_annot = d, g_auc\n                best_state = {k: v.detach().cpu().clone()\n                              for k, v in model.state_dict().items()}\n            if time.time() - T0 > TIME_BUDGET:\n                log(\"  time budget reached\")\n                break\n\n        if best_state is not None:\n            model.load_state_dict(best_state)\n        results[cfg[\"name\"]] = (best, best_annot)\n        test_preds[cfg[\"name\"]] = predict(model, Cte, Mte, np.arange(len(st_te)), dev,\n                                          cfg[\"img\"])\n        log(f\"  {cfg['name']}: best holdout {best:.4f} (annot {best_annot:.4f})\")\n        del model, opt, sched, scaler, best_state\n        gc.collect()\n        if dev.type == \"cuda\":\n            torch.cuda.empty_cache()\n\n    log(\"---- summary ----\")\n    for n, (d, g_auc) in results.items():\n        log(f\"  {n:12s} holdout {d:.4f}   annot {g_auc:.4f}\")\n    pick = max(results, key=lambda k: results[k][0])\n    log(f\"best on the holdout: {pick} ({results[pick][0]:.4f})\")\n\n\n    # ---- write every candidate -------------------------------------------- #\n    # One file per configuration, plus the holdout's choice as `submission.csv`. A run\n    # costs a full decode of the corpus whichever configuration wins, so keeping every\n    # arm makes a later change of configuration free rather than another full run.\n    for name, pred in test_preds.items():\n        sub = write_submission(pred, st_te, test_df, f\"submission_{name}.csv\")\n        log(f\"  submission_{name}.csv {sub.shape}; \"\n            f\"nulls {int(sub[TARGETS].isna().sum().sum())}\")\n\n    ens = np.mean([pd.DataFrame(p).rank(pct=True).values for p in test_preds.values()],\n                  axis=0)\n    write_submission(ens, st_te, test_df, \"submission_rankmean.csv\")\n    log(f\"  submission_rankmean.csv (rank mean of {len(test_preds)})\")\n\n    sub = write_submission(test_preds[pick], st_te, test_df, \"submission.csv\")\n    log(f\"submission.csv = {pick}; {sub.shape}; \"\n        f\"nulls {int(sub[TARGETS].isna().sum().sum())}\")\n    print(sub.head().to_string())\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-10T06:00:34.050367Z","iopub.status.busy":"2026-08-10T06:00:34.049793Z","iopub.status.idle":"2026-08-10T06:00:34.087282Z","shell.execute_reply":"2026-08-10T06:00:34.08677Z"},"papermill":{"duration":0.059827,"end_time":"2026-08-10T06:00:34.088638+00:00","exception":false,"start_time":"2026-08-10T06:00:34.028811+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"8150c020","cell_type":"code","source":"try:\n    main()\nexcept LabelSourceError:\n    # Deliberately not absorbed: see LabelSourceError. A run that trained on the\n    # wrong labels would finish and write a submission worth submitting by mistake.\n    traceback.print_exc()\n    raise\nexcept Exception:\n    traceback.print_exc()\n    # A submission that fails to write scores nothing at all, so fall back to the\n    # benchmark file rather than dying.\n    t = pd.read_csv(find_root() / \"test.csv\")\n    for c in TARGETS:\n        t[c] = 0.5\n    t.to_csv(\"submission.csv\", index=False)\n    print(\"wrote fallback submission.csv\")\nlog(\"done\")\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-10T06:00:34.129122Z","iopub.status.busy":"2026-08-10T06:00:34.128913Z","iopub.status.idle":"2026-08-10T06:01:30.316388Z","shell.execute_reply":"2026-08-10T06:01:30.315803Z"},"papermill":{"duration":56.209211,"end_time":"2026-08-10T06:01:30.318002+00:00","exception":false,"start_time":"2026-08-10T06:00:34.108791+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"8a50740e-d5db-4c50-b964-f08b1a6e14e8","cell_type":"code","source":"# E9: independent RadImageNet ResNet-50 arm for the verified E2 parent.\n#\n# Adapted 2026-08-11 from the V52 cell in the public Kaggle competition notebook\n# prvsiyan/rsna-knee-read-the-report-then-the-knee (latest source SHA-256\n# b54aa529f38dc6f594478e7975d86459ddbff898453a2c2633f4c3be4b909e61).\n# Kaggle's public-code rule deems public Competition Code open-source; Meta Kaggle\n# documents public notebooks under Apache-2.0. Changes here remove the unavailable B3\n# arm, pin the public E2 OOF bundle, add per-target/no-regression gates, and preserve E2\n# byte-for-byte on every failure. This module is appended as the final notebook cell;\n# its imports and DICOM helpers are supplied by the parent notebook.\nSLOTS = [\n    (\"SAG_FS\", \"Sagittal\", None, True),\n    (\"COR_FS\", \"Coronal\", None, True),\n    (\"AX_FS\", \"Axial\", None, True),\n]\nN_SLOT = len(SLOTS)\nCACHE_SLICES = 8\nTIME_BUDGET = 8.72 * 3600\nIMG = CACHE_IMG = 224\n# Match the released RadImageNet model's full-frame pretraining. Setting a crop\n# larger than every acquisition disables the optional physical crop in read_slot.\nCROP_MM = 10_000.0\nSLICE_BAND = (0.12, 0.88)\n# The audited public OOF was produced after the notebook's legacy member group, which\n# left this process-global pixel contract active. Pin it instead of inheriting whichever\n# E2 group happened to run last. This uses public preprocessing code only; no legacy\n# checkpoint or unknown-license asset is attached.\nRULES = dict(RULES_LEGACY)\nTOKEN_DIM = 2048             # official ResNet-50 global-average feature\nHEAD_DIM = 512\nPINNED_HEADS_SHA256 = \"0f465649799ecfbccaac1767844639e7ced44e1bc9babde6e4bac7c5d9b89eaa\"\nPINNED_REMOTE_AUDIT_SHA256 = \"267f948078710d3ca8a6f0de4ce0a5e75e850e1f08d36451549a137d878a6fe8\"\nPINNED_PUBLIC_DIAGNOSTIC_SHA256 = \"0f2f82fb40f0570d6766f73b0d7f51489df6d0faa8fd6e4c0f45bcd6c4c7b283\"\nPINNED_E9B_CONTRACT_SHA256 = \"6777c0a0ba7dd044752fac948752dc39e9ca35b3c280fe74ee6d27a5865d87e7\"\n# E10 repairs E9's censored search: its alpha grid stopped at 0.25 and four of five outer\n# folds selected that ceiling, so the deployed 0.20 was a boundary artifact rather than an\n# optimum. The ladder, the two-source per-target gains and every deployable weight map live\n# in the hash-pinned contract; this cell recomputes the remote half in-kernel before use.\n# The contract also carries the held-out form of E10's own weight choice: selecting the rung\n# on four grouped folds and scoring the fifth picks 0.60 (public) and 0.70 (v15) in all five\n# outer folds, and never picks 0.20. So every deployable rung at or below 0.35 is below what\n# honest selection would choose, which is the answer to \"you tuned on the 58 gold rows\".\nPINNED_E10_CONTRACT_SHA256 = \"219c91f40905181c222e2966b3fed01a96570ddfb64862357d5fd6cad500cd45\"\nE10_CONFIG = \"uniform_035\"\nE10_PRESERVED_TARGETS = [\"Baker's\", \"Fracture\"]\n\n# E11 trains a third arm whose diversity is in the pixels rather than in the weights. The\n# existing arm reads three fat-suppressed slots at full frame; every one of E2's twenty\n# members reads one DINOv2 recipe. Nothing in the portfolio has yet looked at a\n# non-fat-suppressed series, where meniscal and ligament morphology is conventionally read,\n# and nothing has given RadImageNet a physically normalised field of view. E11 changes both:\n# three non-suppressed slots plus one suppressed anchor, cropped to 130 mm, which the parent\n# notebook establishes is below the acquired field of view of 99.6% of series while still\n# containing the joint. It is a training mode only; it never touches the submission.\nARM_MODE = \"e10\"\nE11_SLOTS = [\n    (\"SAG_NOFS\", \"Sagittal\", None, False),\n    (\"COR_NOFS\", \"Coronal\", None, False),\n    (\"AX_NOFS\", \"Axial\", None, False),\n    (\"SAG_FS\", \"Sagittal\", None, True),\n]\nE11_CROP_MM = 130.0\nE11_CACHE_SLICES = 8\nE11_IMG = 224\n# Availability of non-suppressed series per plane is unmeasured, so a fill floor rather than\n# the parent's 90% rule: below this the run has found something structurally wrong, above it\n# an empty slot is simply masked out of the token set like any other absent series.\nE11_MIN_FILL = 0.45\n\n\ndef _v52_as_bool(value):\n    if pd.isna(value):\n        return None\n    text = str(value).strip().upper()\n    if text in {\"1\", \"TRUE\", \"T\", \"YES\", \"Y\"}:\n        return True\n    if text in {\"0\", \"FALSE\", \"F\", \"NO\", \"N\"}:\n        return False\n    try:\n        number = float(text)\n        return True if number == 1 else False if number == 0 else None\n    except Exception:\n        return None\n\n\ndef audit_official_sequence_metadata(inferred, official):\n    \"\"\"Audit metadata agreement without changing the checkpoint pixel contract.\"\"\"\n    needed = {\"SeriesInstanceUID\", \"Fluid_Sensitive\", \"Fat_Suppression\"}\n    if inferred.empty or official.empty or not needed.issubset(official.columns):\n        return\n    inferred_flags = inferred[[\"SeriesInstanceUID\", \"fluid\", \"fatsat\"]].copy()\n    official_flags = official[\n        [\"SeriesInstanceUID\", \"Fluid_Sensitive\", \"Fat_Suppression\"]\n    ].copy()\n    official_flags[\"official_fluid\"] = official_flags[\"Fluid_Sensitive\"].map(\n        _v52_as_bool\n    )\n    official_flags[\"official_fatsat\"] = official_flags[\"Fat_Suppression\"].map(\n        _v52_as_bool\n    )\n    merged = inferred_flags.merge(\n        official_flags[[\"SeriesInstanceUID\", \"official_fluid\", \"official_fatsat\"]],\n        on=\"SeriesInstanceUID\",\n        how=\"inner\",\n    )\n    for inferred_col, official_col, name in [\n        (\"fluid\", \"official_fluid\", \"Fluid_Sensitive\"),\n        (\"fatsat\", \"official_fatsat\", \"Fat_Suppression\"),\n    ]:\n        valid = merged[official_col].notna() & merged[inferred_col].notna()\n        if valid.any():\n            agreement = (\n                merged.loc[valid, inferred_col].astype(bool).to_numpy()\n                == merged.loc[valid, official_col].astype(bool).to_numpy()\n            ).mean()\n            log(\n                f\"V52 metadata audit {name}: {agreement:.1%} agreement \"\n                f\"on {int(valid.sum())} series\"\n            )\n\n\ndef find_input_file(name):\n    for root, dirs, files in os.walk(\"/kaggle/input\"):\n        dirs[:] = [d for d in dirs if d not in (\"train_series\", \"test_series\")]\n        if name in files:\n            return Path(root) / name\n    raise FileNotFoundError(name)\n\n\ndef find_input_dir(name):\n    for root, dirs, files in os.walk(\"/kaggle/input\"):\n        if Path(root).name == name:\n            return Path(root)\n    raise FileNotFoundError(name)\n\n\ndef make_targets(train):\n    \"\"\"Three independent public report teachers; image-read gold always wins.\"\"\"\n    uid = \"StudyInstanceUID\"\n    sources = [\n        pd.read_csv(find_input_file(\"report_labels_v2.csv\")),\n        pd.read_csv(find_input_file(\"llm_labels_v2.csv\")),\n        pd.read_csv(find_input_file(\"labels_llm_gpt56sol.csv\")),\n    ]\n    cube = []\n    for frame in sources:\n        if frame[uid].duplicated().any():\n            raise ValueError(\"duplicate study in report-label source\")\n        aligned = train[[uid]].merge(frame[[uid] + TARGETS], on=uid, how=\"left\")\n        cube.append(aligned[TARGETS].to_numpy(float))\n    cube = np.stack(cube)\n    available = np.isfinite(cube).sum(0)\n    if np.any(available < 2):\n        raise ValueError(\"fewer than two report teachers for a study/target\")\n    y = np.nanmean(cube, axis=0).astype(np.float32)\n    disagreement = np.nanmean(np.abs(cube - y[None]), axis=0)\n    agreement = np.clip(1.0 - 2.0 * disagreement, 0, 1)\n    certainty = np.clip(2.0 * np.abs(y - .5), 0, 1)\n    w = (.15 + .85 * (.65 * agreement + .35 * certainty)).astype(np.float32)\n    gold = train[TARGETS].notna().all(axis=1).to_numpy()\n    y[gold] = train.loc[gold, TARGETS].to_numpy(np.float32)\n    w[gold] = 3.0\n    return y, w, gold\n\n\ndef report_groups(train):\n    report = (train.Report.fillna(\"\").astype(str).str.lower()\n              .str.replace(r\"\\s+\", \" \", regex=True).str.strip())\n    return np.array([hashlib.sha256(x.encode()).hexdigest()[:24] for x in report])\n\n\ndef _v52_sha256(path):\n    digest = hashlib.sha256()\n    with open(path, \"rb\") as handle:\n        for chunk in iter(lambda: handle.read(8 << 20), b\"\"):\n            digest.update(chunk)\n    return digest.hexdigest()\n\n\ndef load_radimagenet(device):\n    \"\"\"Strictly load the official RadImageNet ResNet-50 PyTorch checkpoint.\"\"\"\n    from torchvision.models import resnet50\n\n    checkpoint = find_input_file(\"ResNet50.pt\")\n    expected_checkpoint = \"08629f7e7bd3e29b8ee9522ca3f65ce4d010a7ddf74f0ea3c7e3f3d0bbab0734\"\n    observed_checkpoint = _v52_sha256(checkpoint)\n    if observed_checkpoint != expected_checkpoint:\n        raise RuntimeError(f\"RadImageNet checkpoint drift: {observed_checkpoint}\")\n\n    class RadImageNetEncoder(nn.Module):\n        def __init__(self):\n            super().__init__()\n            self.backbone = nn.Sequential(\n                *list(resnet50(weights=None).children())[:-2]\n            )\n\n        def forward(self, image):\n            return self.backbone(image).mean(dim=(2, 3))\n\n    model = RadImageNetEncoder()\n    state = torch.load(checkpoint, map_location=\"cpu\", weights_only=True)\n    if not state or not all(str(key).startswith(\"backbone.\") for key in state):\n        raise RuntimeError(\"unexpected RadImageNet state-dict namespace\")\n    model.load_state_dict(state, strict=True)\n    parameter_count = sum(parameter.numel() for parameter in model.parameters())\n    if parameter_count != 23_508_032:\n        raise RuntimeError(f\"unexpected RadImageNet parameter count {parameter_count}\")\n    model.eval().to(device)\n    for parameter in model.parameters():\n        parameter.requires_grad_(False)\n    gpu_count = torch.cuda.device_count() if device.type == \"cuda\" else 0\n    if gpu_count > 1:\n        model = nn.DataParallel(model, device_ids=list(range(gpu_count)))\n    log(\n        f\"RadImageNet strict load: {parameter_count:,} params; \"\n        f\"inference GPUs={max(1, gpu_count)}\"\n    )\n    return model\n\n\n@torch.inference_mode()\ndef encode_radimagenet(cache, slot_mask, device):\n    \"\"\"Encode acquired slices with the official [-1, 1] RadImageNet contract.\"\"\"\n    n, slots, slices, h, w = cache.shape\n    features = np.zeros((n, slots * slices, TOKEN_DIM), np.float16)\n    token_mask = np.repeat(slot_mask[:, :, None], slices, axis=2).reshape(n, -1)\n    valid = np.flatnonzero(token_mask.reshape(-1) > 0)\n    flat = cache.reshape(-1, h, w)\n    model = load_radimagenet(device)\n    if device.type == \"cuda\":\n        batch = 192 if torch.cuda.device_count() > 1 else 96\n    else:\n        batch = 8\n    for b0 in range(0, len(valid), batch):\n        ix = valid[b0:b0 + batch]\n        x = torch.from_numpy(flat[ix]).to(device).float().div_(127.5).sub_(1.0)\n        x = x.unsqueeze(1).expand(-1, 3, -1, -1).contiguous()\n        with torch.autocast(\"cuda\", enabled=device.type == \"cuda\"):\n            feat = model(x)\n        if feat.shape[1:] != (TOKEN_DIM,):\n            raise RuntimeError(f\"unexpected RadImageNet feature shape {tuple(feat.shape)}\")\n        features.reshape(-1, TOKEN_DIM)[ix] = (\n            feat.float().cpu().numpy().astype(np.float16)\n        )\n        if b0 % (batch * 100) == 0:\n            log(f\"RadImageNet encoded {b0}/{len(valid)} acquired slices\")\n    del model\n    if device.type == \"cuda\":\n        torch.cuda.empty_cache()\n    return features, token_mask.astype(np.float32)\n\n\nclass FoundationQueryHead(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.project = nn.Sequential(nn.LayerNorm(TOKEN_DIM),\n                                     nn.Linear(TOKEN_DIM, HEAD_DIM), nn.GELU())\n        self.plane = nn.Parameter(torch.randn(N_SLOT, HEAD_DIM) * .01)\n        self.position = nn.Parameter(torch.randn(CACHE_SLICES, HEAD_DIM) * .01)\n        self.query = nn.Parameter(torch.randn(len(TARGETS), HEAD_DIM) * .02)\n        self.attn = nn.MultiheadAttention(HEAD_DIM, 8, dropout=.10, batch_first=True)\n        self.fuse = nn.Sequential(\n            nn.LayerNorm(HEAD_DIM * 4), nn.Linear(HEAD_DIM * 4, HEAD_DIM),\n            nn.GELU(), nn.Dropout(.15),\n        )\n        self.weight = nn.Parameter(torch.randn(len(TARGETS), HEAD_DIM) * .02)\n        self.bias = nn.Parameter(torch.zeros(len(TARGETS)))\n\n    def forward(self, feature, mask):\n        token = self.project(feature.float())\n        token = token.view(len(token), N_SLOT, CACHE_SLICES, HEAD_DIM)\n        token = token + self.plane[None, :, None] + self.position[None, None]\n        token = token.flatten(1, 2)\n        key_padding = mask <= 0\n        # No study should be empty, but keep MHA numerically defined if one is.\n        all_empty = key_padding.all(1)\n        if all_empty.any():\n            key_padding = key_padding.clone()\n            key_padding[all_empty, 0] = False\n        query = self.query.unsqueeze(0).expand(len(token), -1, -1)\n        attended = query + self.attn(query, token, token,\n                                     key_padding_mask=key_padding,\n                                     need_weights=False)[0]\n        denom = mask.sum(1, keepdim=True).clamp_min(1).unsqueeze(-1)\n        mean = (token * mask.unsqueeze(-1)).sum(1, keepdims=True) / denom\n        mean = mean.expand(-1, len(TARGETS), -1)\n        fused = self.fuse(torch.cat(\n            [attended, mean, torch.abs(attended - mean), attended * mean], -1))\n        return (fused * self.weight.unsqueeze(0)).sum(-1) + self.bias\n\n\ndef macro_auc(y, pred):\n    from sklearn.metrics import roc_auc_score\n    hard = (np.asarray(y) >= .5).astype(np.uint8)\n    values = [roc_auc_score(hard[:, j], pred[:, j])\n              for j in range(hard.shape[1]) if np.unique(hard[:, j]).size == 2]\n    return float(np.mean(values))\n\n\ndef _v52_target_auc(y, pred):\n    from sklearn.metrics import roc_auc_score\n    hard = (np.asarray(y) >= .5).astype(np.uint8)\n    return {\n        target: float(roc_auc_score(hard[:, index], pred[:, index]))\n        for index, target in enumerate(TARGETS)\n    }\n\n\n@torch.inference_mode()\ndef predict_head(model, features, masks, indices, device, batch=64):\n    model.eval()\n    pred = []\n    for b0 in range(0, len(indices), batch):\n        ix = indices[b0:b0 + batch]\n        x = torch.from_numpy(features[ix]).to(device)\n        m = torch.from_numpy(masks[ix]).to(device)\n        with torch.autocast(\"cuda\", enabled=device.type == \"cuda\"):\n            pred.append(torch.sigmoid(model(x, m)).float().cpu())\n    return torch.cat(pred).numpy()\n\n\ndef train_fold(features, masks, y, weights, train_idx, val_idx, fold, device):\n    from torch.utils.data import DataLoader, Dataset\n    class Rows(Dataset):\n        def __init__(self, indices): self.indices = np.asarray(indices)\n        def __len__(self): return len(self.indices)\n        def __getitem__(self, k):\n            i = self.indices[k]\n            return features[i], masks[i], y[i], weights[i]\n    model = FoundationQueryHead().to(device)\n    optimizer = torch.optim.AdamW(model.parameters(), lr=2e-4, weight_decay=3e-3)\n    generator = torch.Generator().manual_seed(SEED + 100 + fold)\n    loader = DataLoader(Rows(train_idx), batch_size=48, shuffle=True,\n                        generator=generator, num_workers=2, pin_memory=True,\n                        persistent_workers=True)\n    best, best_auc, stale = None, -1.0, 0\n    for epoch in range(24):\n        model.train()\n        for x, m, target, weight in loader:\n            x, m = x.to(device), m.to(device)\n            target, weight = target.to(device), weight.to(device)\n            with torch.autocast(\"cuda\", enabled=device.type == \"cuda\"):\n                logits = model(x, m)\n                raw = F.binary_cross_entropy_with_logits(logits, target,\n                                                          reduction=\"none\")\n                loss = (raw * weight).sum() / weight.sum().clamp_min(1)\n            optimizer.zero_grad(set_to_none=True)\n            loss.backward()\n            nn.utils.clip_grad_norm_(model.parameters(), 1.0)\n            optimizer.step()\n        pred = predict_head(model, features, masks, val_idx, device)\n        score = macro_auc(y[val_idx], pred)\n        log(f\"fold {fold} epoch {epoch}: grouped weak-val AUC {score:.5f}\")\n        if score > best_auc + 2e-4:\n            best_auc, stale = score, 0\n            best = {k: v.detach().cpu() for k, v in model.state_dict().items()}\n        else:\n            stale += 1\n            if stale >= 5: break\n    return best, best_auc\n\n\ndef _v52_rank_columns(values):\n    frame = pd.DataFrame(np.asarray(values, dtype=np.float64))\n    return frame.rank(method=\"average\", pct=True).to_numpy(np.float64)\n\n\ndef _v52_validate_submission(frame, expected_ids):\n    expected_columns = [\"StudyInstanceUID\", *TARGETS]\n    if frame.columns.tolist() != expected_columns:\n        raise RuntimeError(\"V52 submission schema drift\")\n    ids = frame[\"StudyInstanceUID\"].astype(str).tolist()\n    if ids != list(map(str, expected_ids)) or len(ids) != len(set(ids)):\n        raise RuntimeError(\"V52 submission study identity/order drift\")\n    values = frame[TARGETS].to_numpy(np.float64)\n    if not np.isfinite(values).all() or values.min() < 0 or values.max() > 1:\n        raise RuntimeError(\"V52 submission values are invalid\")\n\n\ndef main_v52():\n    import shutil\n    from sklearn.model_selection import GroupKFold\n\n    output = Path(\"/kaggle/working/rsna_rad_e9\")\n    output.mkdir(parents=True, exist_ok=True)\n    primary = Path(\"/kaggle/working/submission.csv\")\n    preserved = Path(\"/kaggle/working/submission_e2_preserved.csv\")\n    audit_path = Path(\"/kaggle/working/rad_e9_audit.json\")\n    audit = {\n        \"status\": \"E2_PRESERVED\",\n        \"evidence_boundary\": (\n            \"All OOF values are local diagnostics on 58 official image labels; \"\n            \"they are not Kaggle competition scores. E2 remains the primary unless \"\n            \"strict artifact, OOF, inference, and submission gates all pass.\"\n        ),\n        \"encoder\": \"RadImageNet ResNet-50 official PyTorch release\",\n        \"encoder_license\": \"CC-BY-NC-SA-4.0 (Kaggle-hosted weight metadata)\",\n        \"encoder_sha256\": \"08629f7e7bd3e29b8ee9522ca3f65ce4d010a7ddf74f0ea3c7e3f3d0bbab0734\",\n        \"encoder_source_commit\": \"0ce16f7375db4236e646829d1eca61cdb4282133\",\n        \"base_oof_sha256\": \"62d47ba4c0c8347b5b24e7fd2aa517aae0fd6d4656fd5d829fd3a50f0159909c\",\n        \"parent\": \"E2 captured 20-member DINOv2 rank ensemble\",\n        \"blend_contract\": \"rank columns independently, then 80% E2 plus 20% RadImageNet\",\n        \"pixel_rules\": dict(RULES),\n    }\n    if not primary.is_file():\n        raise FileNotFoundError(\"E2 parent submission is absent\")\n    shutil.copy2(primary, preserved)\n\n    try:\n        device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\n        if device.type != \"cuda\":\n            raise RuntimeError(\"V52 RadImageNet experiment requires CUDA\")\n        elapsed = max(0.0, time.time() - float(globals().get(\"T0\", time.time())))\n        available = 8.72 * 3600 - elapsed\n        audit[\"elapsed_before_v52_seconds\"] = elapsed\n        audit[\"available_at_start_seconds\"] = available\n        if available < 2.0 * 3600:\n            raise TimeoutError(f\"only {available / 60:.1f} minutes remain\")\n\n        train = pd.read_csv(ROOT / \"train.csv\", dtype={\"StudyInstanceUID\": str})\n        train_series = pd.read_csv(\n            ROOT / \"train_series.csv\",\n            dtype={\"StudyInstanceUID\": str, \"SeriesInstanceUID\": str},\n        )\n        if len(train) != 4407:\n            raise RuntimeError(f\"unexpected train study count {len(train)}\")\n        plane = dict(zip(train_series.SeriesInstanceUID, train_series.Anatomical_Plane))\n        headers = annotate(walk(\"train_series\"))\n        audit_official_sequence_metadata(headers, train_series)\n        studies, pixels, slot_mask = build_cache(\n            pick_slots(headers, plane), plane, lat_of(headers, \"train-v52 \"), \"train-v52\"\n        )\n        by_uid = {str(uid): i for i, uid in enumerate(studies)}\n        missing = [uid for uid in train.StudyInstanceUID if uid not in by_uid]\n        if missing:\n            raise RuntimeError(f\"{len(missing)} train studies absent from cache\")\n        order = np.array([by_uid[uid] for uid in train.StudyInstanceUID], dtype=np.int64)\n        pixels, slot_mask = pixels[order], slot_mask[order]\n        train_token_count = int(np.repeat(slot_mask[:, :, None], CACHE_SLICES, 2).sum())\n        if train_token_count < int(0.90 * len(train) * N_SLOT * CACHE_SLICES):\n            raise RuntimeError(f\"insufficient acquired train slices: {train_token_count}\")\n        features, token_mask = encode_radimagenet(pixels, slot_mask, device)\n        del pixels, slot_mask, headers\n        gc.collect()\n\n        y, weights, gold = make_targets(train)\n        if int(gold.sum()) != 58:\n            raise RuntimeError(f\"expected 58 fully gold studies, observed {int(gold.sum())}\")\n        groups = report_groups(train)\n        if len(np.unique(groups)) < 4000:\n            raise RuntimeError(\"unexpected report-group collapse\")\n\n        splits = list(GroupKFold(5).split(features, groups=groups))\n        fold_id = np.full(len(train), -1, dtype=np.int8)\n        folds = []\n        oof = np.zeros_like(y, dtype=np.float32)\n        for fold, (tr, va) in enumerate(splits):\n            if set(groups[tr]).intersection(groups[va]):\n                raise RuntimeError(f\"report leakage in fold {fold}\")\n            fold_id[va] = fold\n            state, score = train_fold(\n                features, token_mask, y, weights, tr, va, fold, device\n            )\n            if state is None:\n                raise RuntimeError(f\"fold {fold} produced no checkpoint\")\n            head = FoundationQueryHead().to(device)\n            head.load_state_dict(state, strict=True)\n            oof[va] = predict_head(head, features, token_mask, va, device)\n            folds.append({\"fold\": fold, \"weak_auc\": float(score), \"state_dict\": state})\n            del head\n            torch.cuda.empty_cache()\n        if (fold_id < 0).any() or not np.isfinite(oof).all():\n            raise RuntimeError(\"incomplete V52 OOF\")\n\n        weak_auc = macro_auc(y, oof)\n        gold_auc = macro_auc(y[gold], oof[gold])\n        log(f\"V52 RadImageNet OOF weak macro AUC {weak_auc:.5f}\")\n        log(f\"V52 RadImageNet OOF gold macro AUC {gold_auc:.5f} on 58 studies\")\n        torch.save(\n            {\n                \"version\": \"v52-radimagenet-resnet50-official-1\",\n                \"targets\": TARGETS,\n                \"encoder_sha256\": audit[\"encoder_sha256\"],\n                \"encoder_source_commit\": audit[\"encoder_source_commit\"],\n                \"img\": IMG,\n                \"slices_per_plane\": CACHE_SLICES,\n                \"feature\": \"global_average_pool\",\n                \"folds\": folds,\n                \"weak_oof_auc\": weak_auc,\n                \"gold_oof_auc\": gold_auc,\n            },\n            output / \"v52_radimagenet_heads.pt\",\n        )\n        oof_frame = pd.DataFrame(oof, columns=TARGETS)\n        oof_frame.insert(0, \"StudyInstanceUID\", train.StudyInstanceUID)\n        oof_frame[\"fold\"] = fold_id\n        oof_frame[\"is_gold\"] = gold.astype(np.uint8)\n        oof_frame.to_csv(output / \"v52_oof.csv\", index=False)\n\n        base_npz = find_input_file(\"oof.npz\")\n        observed_base_hash = _v52_sha256(base_npz)\n        if observed_base_hash != audit[\"base_oof_sha256\"]:\n            raise RuntimeError(f\"E2 OOF artifact drift: {observed_base_hash}\")\n        with np.load(base_npz, allow_pickle=False) as base_bundle:\n            expected_members = {\"ids\", \"pred\", \"y_derived\", \"gold_mask\", \"targets\"}\n            if set(base_bundle.files) != expected_members:\n                raise RuntimeError(f\"unexpected E2 OOF members: {base_bundle.files}\")\n            base_ids = base_bundle[\"ids\"].astype(str)\n            base_targets = base_bundle[\"targets\"].astype(str).tolist()\n            base_gold = base_bundle[\"gold_mask\"].astype(bool)\n            base_prediction = base_bundle[\"pred\"].astype(np.float64)\n        if base_targets != TARGETS:\n            raise RuntimeError(\"E2 OOF target order drift\")\n        if not np.array_equal(base_ids, train.StudyInstanceUID.astype(str).to_numpy()):\n            raise RuntimeError(\"E2 OOF study order drift\")\n        if not np.array_equal(base_gold, gold):\n            raise RuntimeError(\"E2 OOF gold mask differs from official train.csv\")\n        train_rows = np.flatnonzero(gold)\n        if len(train_rows) != 58:\n            raise RuntimeError(f\"expected 58 E2 gold rows, observed {len(train_rows)}\")\n        gold_y = train.loc[gold, TARGETS].to_numpy(np.float64)\n        exact_public = base_prediction[gold]\n        rad = oof[gold].astype(np.float64)\n        if not all(np.isfinite(x).all() for x in (gold_y, exact_public, rad)):\n            raise RuntimeError(\"non-finite aligned E2/RadImageNet OOF value\")\n\n        base_rank = _v52_rank_columns(exact_public)\n        rad_rank = _v52_rank_columns(rad)\n        base_score = macro_auc(gold_y, base_rank)\n        rad_score = macro_auc(gold_y, rad_rank)\n        alpha_grid = np.array([0.0, 0.025, 0.05, 0.10, 0.15, 0.20, 0.25])\n        gold_folds = fold_id[train_rows]\n        if sorted(np.unique(gold_folds).tolist()) != [0, 1, 2, 3, 4]:\n            raise RuntimeError(\"gold rows do not cover all five grouped folds\")\n        nested = np.zeros_like(base_rank)\n        choices = []\n        outer_train_scores = []\n        for outer in range(5):\n            tr = gold_folds != outer\n            va = ~tr\n            scored = []\n            for alpha in alpha_grid:\n                blend = (1.0 - alpha) * base_rank[tr] + alpha * rad_rank[tr]\n                score = macro_auc(gold_y[tr], blend) - 0.01 * float(alpha)\n                scored.append(float(score))\n            best = max(range(len(alpha_grid)), key=lambda i: (scored[i], -alpha_grid[i]))\n            alpha = float(alpha_grid[best])\n            choices.append(alpha)\n            outer_train_scores.append(scored)\n            nested[va] = (1.0 - alpha) * base_rank[va] + alpha * rad_rank[va]\n        nested_score = macro_auc(gold_y, nested)\n        # Deployment weight is fixed by the independently scored public 0.906 mechanism.\n        # The 58 gold rows may veto it and measure fold stability, but do not tune it.\n        final_alpha = 0.20\n        final_oof = (1.0 - final_alpha) * base_rank + final_alpha * rad_rank\n        final_score = macro_auc(gold_y, final_oof)\n        grid_scores = {\n            f\"{alpha:.3f}\": macro_auc(\n                gold_y, (1.0 - alpha) * base_rank + alpha * rad_rank\n            )\n            for alpha in alpha_grid\n        }\n        base_target_scores = _v52_target_auc(gold_y, base_rank)\n        rad_target_scores = _v52_target_auc(gold_y, rad_rank)\n        final_target_scores = _v52_target_auc(gold_y, final_oof)\n        target_deltas = {\n            target: final_target_scores[target] - base_target_scores[target]\n            for target in TARGETS\n        }\n        target_regressions = {\n            target: delta for target, delta in target_deltas.items() if delta < -1e-12\n        }\n        positive_folds = int(sum(alpha > 0 for alpha in choices))\n        supported = bool(\n            final_alpha > 0\n            and positive_folds >= 3\n            and nested_score >= base_score + 0.001\n            and final_score >= base_score + 0.001\n            and not target_regressions\n        )\n        audit[\"oof\"] = {\n            \"rows\": 58,\n            \"weak_macro_auc\": weak_auc,\n            \"rad_gold_macro_auc\": rad_score,\n            \"e2_macro_auc\": base_score,\n            \"outer_fold_choices\": choices,\n            \"outer_fold_penalized_train_scores\": outer_train_scores,\n            \"nested_blend_macro_auc\": nested_score,\n            \"final_alpha\": final_alpha,\n            \"final_descriptive_macro_auc\": final_score,\n            \"full_grid_macro_auc\": grid_scores,\n            \"positive_outer_folds\": positive_folds,\n            \"per_target\": {\n                target: {\n                    \"e2_auc\": base_target_scores[target],\n                    \"radimagenet_auc\": rad_target_scores[target],\n                    \"blend_auc\": final_target_scores[target],\n                    \"blend_delta\": target_deltas[target],\n                }\n                for target in TARGETS\n            },\n            \"target_regressions\": target_regressions,\n            \"gold_fold_counts\": {\n                str(fold): int((gold_folds == fold).sum()) for fold in range(5)\n            },\n            \"selection_supported\": supported,\n        }\n        audit[\"train_available_slice_tokens\"] = train_token_count\n        audit[\"head_count\"] = len(folds)\n        if not supported:\n            audit[\"status\"] = \"OOF_REJECTED_E2_PRESERVED\"\n            log(\n                f\"V52 rejected by nested OOF: base={base_score:.5f}, \"\n                f\"nested={nested_score:.5f}, final={final_score:.5f}, choices={choices}\"\n            )\n            return\n\n        del features, token_mask\n        gc.collect()\n        test = pd.read_csv(ROOT / \"test.csv\", dtype={\"StudyInstanceUID\": str})\n        test_series = pd.read_csv(\n            ROOT / \"test_series.csv\",\n            dtype={\"StudyInstanceUID\": str, \"SeriesInstanceUID\": str},\n        )\n        test_plane = dict(zip(test_series.SeriesInstanceUID, test_series.Anatomical_Plane))\n        test_headers = annotate(walk(\"test_series\"))\n        audit_official_sequence_metadata(test_headers, test_series)\n        test_studies, test_pixels, test_slot_mask = build_cache(\n            pick_slots(test_headers, test_plane),\n            test_plane,\n            lat_of(test_headers, \"test-v52 \"),\n            \"test-v52\",\n        )\n        test_by_uid = {str(uid): i for i, uid in enumerate(test_studies)}\n        test_missing = [uid for uid in test.StudyInstanceUID if uid not in test_by_uid]\n        if test_missing:\n            raise RuntimeError(f\"{len(test_missing)} test studies absent from cache\")\n        test_order = np.array([test_by_uid[uid] for uid in test.StudyInstanceUID])\n        test_pixels = test_pixels[test_order]\n        test_slot_mask = test_slot_mask[test_order]\n        test_token_count = int(\n            np.repeat(test_slot_mask[:, :, None], CACHE_SLICES, 2).sum()\n        )\n        if test_token_count < int(0.85 * len(test) * N_SLOT * CACHE_SLICES):\n            raise RuntimeError(f\"insufficient acquired test slices: {test_token_count}\")\n        test_features, test_token_mask = encode_radimagenet(\n            test_pixels, test_slot_mask, device\n        )\n        del test_pixels, test_slot_mask, test_headers\n        gc.collect()\n\n        fold_predictions = []\n        all_test = np.arange(len(test), dtype=np.int64)\n        for record in folds:\n            head = FoundationQueryHead().to(device)\n            head.load_state_dict(record[\"state_dict\"], strict=True)\n            fold_predictions.append(\n                predict_head(head, test_features, test_token_mask, all_test, device)\n            )\n            del head\n            torch.cuda.empty_cache()\n        if len(fold_predictions) != 5:\n            raise RuntimeError(\"test inference did not use all five heads\")\n        rad_test = np.mean(np.stack(fold_predictions), axis=0)\n        if not np.isfinite(rad_test).all():\n            raise RuntimeError(\"non-finite RadImageNet test prediction\")\n\n        baseline = pd.read_csv(preserved, dtype={\"StudyInstanceUID\": str})\n        _v52_validate_submission(baseline, test.StudyInstanceUID)\n        rad_frame = pd.DataFrame(rad_test, columns=TARGETS)\n        rad_frame.insert(0, \"StudyInstanceUID\", test.StudyInstanceUID)\n        _v52_validate_submission(rad_frame, test.StudyInstanceUID)\n        rad_frame.to_csv(output / \"submission_rad_only.csv\", index=False)\n        baseline_rank = _v52_rank_columns(baseline[TARGETS].to_numpy())\n        rad_test_rank = _v52_rank_columns(rad_test)\n        selected_path = None\n        for alpha in alpha_grid[1:]:\n            candidate = baseline.copy()\n            candidate[TARGETS] = (\n                (1.0 - alpha) * baseline_rank + alpha * rad_test_rank\n            )\n            _v52_validate_submission(candidate, test.StudyInstanceUID)\n            path = output / f\"submission_e2_rad_{int(round(1000 * alpha)):03d}.csv\"\n            candidate.to_csv(path, index=False)\n            if abs(float(alpha) - final_alpha) < 1e-12:\n                selected_path = path\n        if selected_path is None or not selected_path.is_file():\n            raise RuntimeError(\"selected V52 blend artifact is absent\")\n        selected = pd.read_csv(selected_path, dtype={\"StudyInstanceUID\": str})\n        _v52_validate_submission(selected, test.StudyInstanceUID)\n        audit[\"test_studies\"] = len(test)\n        audit[\"test_available_slice_tokens\"] = test_token_count\n        audit[\"test_head_count\"] = len(fold_predictions)\n        audit[\"selected_path\"] = str(selected_path)\n        audit[\"selected_sha256\"] = _v52_sha256(selected_path)\n        audit[\"fallback_sha256\"] = _v52_sha256(preserved)\n        shutil.copy2(selected_path, primary)\n        if _v52_sha256(primary) != audit[\"selected_sha256\"]:\n            raise RuntimeError(\"primary V52 copy hash mismatch\")\n        audit[\"status\"] = \"CANDIDATE_SELECTED\"\n        log(\n            f\"E9 selected alpha={final_alpha:.3f}; \"\n            f\"nested={nested_score:.5f} vs E2 OOF={base_score:.5f}\"\n        )\n    except Exception as error:\n        audit[\"status\"] = \"ERROR_E2_PRESERVED\"\n        audit[\"error\"] = f\"{type(error).__name__}: {error}\"\n        audit[\"traceback\"] = traceback.format_exc()\n        log(f\"E9 preserves E2: {audit['error']}\")\n    finally:\n        if audit.get(\"status\") != \"CANDIDATE_SELECTED\" and preserved.is_file():\n            shutil.copy2(preserved, primary)\n        audit[\"primary_sha256\"] = _v52_sha256(primary) if primary.is_file() else None\n        audit_path.write_text(json.dumps(audit, indent=2, sort_keys=True) + \"\\n\")\n\n\ndef _v52_load_pinned_e9b():\n    \"\"\"Load the public v15 heads and reconstruct the dual-OOF target gate.\"\"\"\n    heads_path = find_input_file(\"v52_radimagenet_heads.pt\")\n    remote_path = find_input_file(\"rad_e9_audit.json\")\n    public_path = find_input_file(\"public_oof_diagnostic.json\")\n    contract_path = find_input_file(\"e9b_contract.json\")\n    expected_hashes = {\n        heads_path: PINNED_HEADS_SHA256,\n        remote_path: PINNED_REMOTE_AUDIT_SHA256,\n        public_path: PINNED_PUBLIC_DIAGNOSTIC_SHA256,\n        contract_path: PINNED_E9B_CONTRACT_SHA256,\n    }\n    for path, expected in expected_hashes.items():\n        observed = _v52_sha256(path)\n        if observed != expected:\n            raise RuntimeError(f\"pinned E9b artifact drift for {path.name}: {observed}\")\n\n    remote = json.loads(remote_path.read_text())\n    public = json.loads(public_path.read_text())\n    contract = json.loads(contract_path.read_text())\n    if remote.get(\"status\") != \"OOF_REJECTED_E2_PRESERVED\":\n        raise RuntimeError(f\"unexpected v15 audit status {remote.get('status')}\")\n    if remote.get(\"primary_sha256\") != (\n        \"f9fb57b7bac8489a5d5285b3984b06df57f142572be6417eac6341c43e96707a\"\n    ):\n        raise RuntimeError(\"v15 did not preserve the exact E2 visible artifact\")\n    remote_oof = remote.get(\"oof\", {})\n    public_target = public.get(\"per_target\", {})\n    remote_target = remote_oof.get(\"per_target\", {})\n    if set(public_target) != set(TARGETS) or set(remote_target) != set(TARGETS):\n        raise RuntimeError(\"E9b diagnostic target set drift\")\n    if int(remote.get(\"head_count\", -1)) != 5:\n        raise RuntimeError(\"v15 remote audit does not contain five heads\")\n    if int(remote_oof.get(\"positive_outer_folds\", -1)) != 5:\n        raise RuntimeError(\"v15 remote outer-fold support drift\")\n    if int(public.get(\"positive_outer_folds\", -1)) != 5:\n        raise RuntimeError(\"public outer-fold support drift\")\n\n    selected_targets = [\n        target for target in TARGETS\n        if float(public_target[target][\"blend_delta\"]) > 1e-12\n        and float(remote_target[target][\"blend_delta\"]) > 1e-12\n    ]\n    preserved_targets = [target for target in TARGETS if target not in selected_targets]\n    if selected_targets != contract.get(\"selected_targets\"):\n        raise RuntimeError(f\"E9b selected-target contract drift: {selected_targets}\")\n    if preserved_targets != contract.get(\"preserved_targets\"):\n        raise RuntimeError(f\"E9b preserved-target contract drift: {preserved_targets}\")\n    if len(selected_targets) != 10 or preserved_targets != [\"Baker's\", \"Fracture\"]:\n        raise RuntimeError(\"E9b requires the ten-target dual-OOF intersection\")\n    alpha = float(contract.get(\"alpha\", -1))\n    if abs(alpha - 0.20) > 1e-12:\n        raise RuntimeError(f\"E9b alpha drift: {alpha}\")\n\n    def selective_macro(records, base_key, blend_key):\n        return float(np.mean([\n            float(records[target][blend_key] if target in selected_targets\n                  else records[target][base_key])\n            for target in TARGETS\n        ]))\n\n    public_base = float(public[\"base_gold_macro_auc\"])\n    remote_base = float(remote_oof[\"e2_macro_auc\"])\n    public_selective = selective_macro(public_target, \"base_auc\", \"blend_auc\")\n    remote_selective = selective_macro(remote_target, \"e2_auc\", \"blend_auc\")\n    if public_selective < public_base + 0.001:\n        raise RuntimeError(\"E9b public selective gate no longer improves E2\")\n    if remote_selective < remote_base + 0.001:\n        raise RuntimeError(\"E9b remote selective gate no longer improves E2\")\n\n    payload = torch.load(heads_path, map_location=\"cpu\", weights_only=True)\n    expected_payload = {\n        \"version\": \"v52-radimagenet-resnet50-official-1\",\n        \"targets\": TARGETS,\n        \"encoder_sha256\": (\n            \"08629f7e7bd3e29b8ee9522ca3f65ce4d010a7ddf74f0ea3c7e3f3d0bbab0734\"\n        ),\n        \"encoder_source_commit\": \"0ce16f7375db4236e646829d1eca61cdb4282133\",\n        \"img\": 224,\n        \"slices_per_plane\": 8,\n        \"feature\": \"global_average_pool\",\n    }\n    for key, expected in expected_payload.items():\n        if payload.get(key) != expected:\n            raise RuntimeError(f\"pinned E9b head contract drift for {key}\")\n    folds = payload.get(\"folds\")\n    if not isinstance(folds, list) or len(folds) != 5:\n        raise RuntimeError(\"pinned E9b payload requires five folds\")\n    if sorted(int(record.get(\"fold\", -1)) for record in folds) != list(range(5)):\n        raise RuntimeError(\"pinned E9b fold identity drift\")\n    if any(not isinstance(record.get(\"state_dict\"), dict) for record in folds):\n        raise RuntimeError(\"pinned E9b state dictionary is absent\")\n    if abs(float(payload.get(\"weak_oof_auc\", -1)) - 0.8278261335825697) > 1e-12:\n        raise RuntimeError(\"pinned E9b weak OOF drift\")\n    if abs(float(payload.get(\"gold_oof_auc\", -1)) - 0.8543239133509962) > 1e-12:\n        raise RuntimeError(\"pinned E9b gold OOF drift\")\n    return {\n        \"payload\": payload,\n        \"folds\": folds,\n        \"alpha\": alpha,\n        \"selected_targets\": selected_targets,\n        \"preserved_targets\": preserved_targets,\n        \"public_base\": public_base,\n        \"public_selective\": public_selective,\n        \"remote_base\": remote_base,\n        \"remote_selective\": remote_selective,\n        \"remote_outer_fold_choices\": remote_oof[\"outer_fold_choices\"],\n    }\n\n\ndef _v52_e10_remote_ladder(contract):\n    \"\"\"Recompute this account's half of the ladder from artifacts the kernel can read.\n\n    The contract carries per-target gains for two independent RadImageNet OOF runs. Only the\n    public run is unverifiable here, so its numbers stay data. The remote run is rebuilt from\n    the attached OOF table, the pinned E2 OOF bundle and the official labels, and must match\n    the contract exactly or E10 refuses to deploy.\n    \"\"\"\n    train = pd.read_csv(ROOT / \"train.csv\", dtype={\"StudyInstanceUID\": str})\n    gold = train[TARGETS].notna().all(axis=1).to_numpy()\n    oof_path = find_input_file(\"v52_oof.csv\")\n    observed = _v52_sha256(oof_path)\n    if observed != contract[\"remote_oof_sha256\"]:\n        raise RuntimeError(f\"E10 remote OOF drift: {observed}\")\n    rad_frame = pd.read_csv(oof_path, dtype={\"StudyInstanceUID\": str})\n    if rad_frame.columns.tolist() != [\"StudyInstanceUID\", *TARGETS, \"fold\", \"is_gold\"]:\n        raise RuntimeError(\"E10 remote OOF schema drift\")\n    aligned = train[[\"StudyInstanceUID\"]].merge(\n        rad_frame, on=\"StudyInstanceUID\", how=\"left\", validate=\"one_to_one\"\n    )\n    if aligned[TARGETS].isna().any().any():\n        raise RuntimeError(\"E10 remote OOF does not cover every official train study\")\n\n    base_npz = find_input_file(\"oof.npz\")\n    with np.load(base_npz, allow_pickle=False) as bundle:\n        if bundle[\"targets\"].astype(str).tolist() != TARGETS:\n            raise RuntimeError(\"E10 E2 OOF target order drift\")\n        if not np.array_equal(\n            bundle[\"ids\"].astype(str), train.StudyInstanceUID.astype(str).to_numpy()\n        ):\n            raise RuntimeError(\"E10 E2 OOF study order drift\")\n        if not np.array_equal(bundle[\"gold_mask\"].astype(bool), gold):\n            raise RuntimeError(\"E10 E2 gold mask differs from official train.csv\")\n        base_prediction = bundle[\"pred\"].astype(np.float64)\n\n    # Rank within the scored rows, matching both the E9b parent and test-time deployment\n    # where the ranked population and the scored population are the same studies.\n    base = _v52_rank_columns(base_prediction[gold])\n    rad = _v52_rank_columns(aligned[TARGETS].to_numpy(np.float64)[gold])\n    gold_y = train.loc[gold, TARGETS].to_numpy(np.float64)\n    if len(gold_y) != 58 or not np.isfinite(base).all() or not np.isfinite(rad).all():\n        raise RuntimeError(\"E10 gold alignment is incomplete or non-finite\")\n    reference = _v52_target_auc(gold_y, base)\n    if abs(\n        float(np.mean([reference[t] for t in TARGETS]))\n        - float(contract[\"base_gold_macro_auc\"])\n    ) > 1e-9:\n        raise RuntimeError(\"E10 base gold diagnostic drift\")\n\n    rebuilt = {}\n    for key in contract[\"ladder\"]:\n        alpha = float(key)\n        scores = _v52_target_auc(gold_y, (1.0 - alpha) * base + alpha * rad)\n        rebuilt[key] = {t: scores[t] - reference[t] for t in TARGETS}\n    pinned_remote = contract[\"per_target_ladder_delta\"][\"remote_v15\"]\n    if set(rebuilt) != set(pinned_remote):\n        raise RuntimeError(\"E10 ladder key drift\")\n    for key, deltas in rebuilt.items():\n        for target, delta in deltas.items():\n            if abs(delta - float(pinned_remote[key][target])) > 1e-9:\n                raise RuntimeError(\n                    f\"E10 recomputed remote gain disagrees at {key}/{target}: {delta}\"\n                )\n    if any(abs(reference[t] - float(contract[\"per_target_base_auc\"][t])) > 1e-9 for t in TARGETS):\n        raise RuntimeError(\"E10 per-target base AUC drift\")\n    return rebuilt, reference\n\n\ndef _v52_load_e10():\n    \"\"\"Validate the E10 contract, then return the weight map the kernel will deploy.\"\"\"\n    heads_path = find_input_file(\"v52_radimagenet_heads.pt\")\n    remote_path = find_input_file(\"rad_e9_audit.json\")\n    contract_path = find_input_file(\"e10_contract.json\")\n    for path, expected in (\n        (heads_path, PINNED_HEADS_SHA256),\n        (remote_path, PINNED_REMOTE_AUDIT_SHA256),\n        (contract_path, PINNED_E10_CONTRACT_SHA256),\n    ):\n        observed = _v52_sha256(path)\n        if observed != expected:\n            raise RuntimeError(f\"pinned E10 artifact drift for {path.name}: {observed}\")\n\n    contract = json.loads(contract_path.read_text())\n    if contract.get(\"version\") != \"e10-alpha-ladder-2\":\n        raise RuntimeError(f\"unexpected E10 contract version {contract.get('version')}\")\n    if contract.get(\"targets\") != TARGETS:\n        raise RuntimeError(\"E10 contract target order drift\")\n    remote = json.loads(remote_path.read_text())\n    if remote.get(\"status\") != \"OOF_REJECTED_E2_PRESERVED\":\n        raise RuntimeError(f\"unexpected v15 audit status {remote.get('status')}\")\n    if remote.get(\"primary_sha256\") != (\n        \"f9fb57b7bac8489a5d5285b3984b06df57f142572be6417eac6341c43e96707a\"\n    ):\n        raise RuntimeError(\"v15 did not preserve the exact E2 visible artifact\")\n    if int(remote.get(\"head_count\", -1)) != 5:\n        raise RuntimeError(\"v15 remote audit does not contain five heads\")\n\n    rebuilt, base_auc = _v52_e10_remote_ladder(contract)\n    public_ladder = contract[\"per_target_ladder_delta\"][\"public\"]\n    configuration = contract[\"configurations\"].get(E10_CONFIG)\n    if configuration is None:\n        raise RuntimeError(f\"E10 contract has no configuration {E10_CONFIG!r}\")\n    alpha_map = {t: float(configuration[\"alpha_map\"][t]) for t in TARGETS}\n    if any(alpha < 0.0 or alpha > 1.0 for alpha in alpha_map.values()):\n        raise RuntimeError(\"E10 weight outside the unit interval\")\n    preserved = sorted(t for t, alpha in alpha_map.items() if alpha == 0.0)\n    if preserved != sorted(E10_PRESERVED_TARGETS):\n        raise RuntimeError(f\"E10 preserved-target drift: {preserved}\")\n\n    # Two-tier gate. The scored objective is macro AUC, so the binding requirement is that\n    # the deployed map raise the macro in BOTH independent runs -- the public numbers as\n    # pinned data, this account's numbers as recomputed above. A per-target \"never harm any\n    # single label\" rule is strictly stronger than that objective and would veto rungs that\n    # trade a small loss on one label for a large gain on another, so it is enforced only for\n    # configurations that actually claim it. Whichever claim the contract makes is verified;\n    # a configuration cannot quietly assert dual-positivity it no longer has.\n    claims_dual_positive = bool(configuration[\"all_dual_positive\"])\n    observed_dual_positive = True\n    for target, alpha in alpha_map.items():\n        if alpha == 0.0:\n            continue\n        key = f\"{alpha:.2f}\"\n        if key not in rebuilt:\n            raise RuntimeError(f\"E10 weight {key} is outside the audited ladder\")\n        gains = (float(public_ladder[key][target]), float(rebuilt[key][target]))\n        if not all(gain > 0 for gain in gains):\n            observed_dual_positive = False\n            if claims_dual_positive:\n                raise RuntimeError(\n                    f\"E10 dual-source gate rejects {target} at {key}: {gains}\"\n                )\n    if observed_dual_positive != claims_dual_positive:\n        raise RuntimeError(\n            f\"E10 contract claims all_dual_positive={claims_dual_positive} for \"\n            f\"{E10_CONFIG!r} but recomputation observes {observed_dual_positive}\"\n        )\n\n    macro = {}\n    for name, ladder in ((\"public\", public_ladder), (\"remote_v15\", rebuilt)):\n        total = 0.0\n        for target, alpha in alpha_map.items():\n            gain = 0.0 if alpha == 0.0 else float(ladder[f\"{alpha:.2f}\"][target])\n            total += float(base_auc[target]) + gain\n        macro[name] = total / len(TARGETS)\n        if macro[name] <= float(contract[\"base_gold_macro_auc\"]):\n            raise RuntimeError(\n                f\"E10 macro gate rejects {E10_CONFIG!r}: {name} macro {macro[name]} \"\n                f\"does not beat base {contract['base_gold_macro_auc']}\"\n            )\n        if abs(macro[name] - float(configuration[\"descriptive_macro\"][name])) > 1e-9:\n            raise RuntimeError(\n                f\"E10 recomputed {name} macro {macro[name]} disagrees with the contract \"\n                f\"value {configuration['descriptive_macro'][name]}\"\n            )\n\n    payload = torch.load(heads_path, map_location=\"cpu\", weights_only=True)\n    expected_payload = {\n        \"version\": \"v52-radimagenet-resnet50-official-1\",\n        \"targets\": TARGETS,\n        \"encoder_sha256\": (\n            \"08629f7e7bd3e29b8ee9522ca3f65ce4d010a7ddf74f0ea3c7e3f3d0bbab0734\"\n        ),\n        \"encoder_source_commit\": \"0ce16f7375db4236e646829d1eca61cdb4282133\",\n        \"img\": 224,\n        \"slices_per_plane\": 8,\n        \"feature\": \"global_average_pool\",\n    }\n    for key, expected in expected_payload.items():\n        if payload.get(key) != expected:\n            raise RuntimeError(f\"pinned E10 head contract drift for {key}\")\n    folds = payload.get(\"folds\")\n    if not isinstance(folds, list) or len(folds) != 5:\n        raise RuntimeError(\"pinned E10 payload requires five folds\")\n    if sorted(int(record.get(\"fold\", -1)) for record in folds) != list(range(5)):\n        raise RuntimeError(\"pinned E10 fold identity drift\")\n    if any(not isinstance(record.get(\"state_dict\"), dict) for record in folds):\n        raise RuntimeError(\"pinned E10 state dictionary is absent\")\n    if abs(float(payload.get(\"gold_oof_auc\", -1)) - 0.8543239133509962) > 1e-12:\n        raise RuntimeError(\"pinned E10 gold OOF drift\")\n    return {\n        \"payload\": payload,\n        \"folds\": folds,\n        \"contract\": contract,\n        \"configuration\": E10_CONFIG,\n        \"alpha_map\": alpha_map,\n        \"preserved_targets\": sorted(E10_PRESERVED_TARGETS),\n        \"diagnostic_macro\": configuration[\"diagnostic_macro\"],\n        \"recomputed_macro\": macro,\n        \"all_dual_positive\": claims_dual_positive,\n        \"rationale\": configuration[\"rationale\"],\n        \"recomputed_remote_ladder\": rebuilt,\n    }\n\n\ndef main_v52_pinned_e9b():\n    \"\"\"Inference-only E9b from hash-pinned v15 heads; preserve E2 on any failure.\"\"\"\n    import shutil\n\n    output = Path(\"/kaggle/working/rsna_rad_e9b\")\n    output.mkdir(parents=True, exist_ok=True)\n    primary = Path(\"/kaggle/working/submission.csv\")\n    preserved = Path(\"/kaggle/working/submission_e2_preserved.csv\")\n    audit_path = Path(\"/kaggle/working/rad_e9b_audit.json\")\n    audit = {\n        \"status\": \"E2_PRESERVED\",\n        \"mode\": \"pinned_v15_heads_inference_only\",\n        \"evidence_boundary\": (\n            \"OOF values are diagnostics, not Kaggle scores. The fixed public 20-percent \"\n            \"vote is applied only to targets improving in two independent OOF runs.\"\n        ),\n        \"encoder\": \"RadImageNet ResNet-50 official PyTorch release\",\n        \"encoder_license\": \"CC-BY-NC-SA-4.0\",\n        \"encoder_sha256\": (\n            \"08629f7e7bd3e29b8ee9522ca3f65ce4d010a7ddf74f0ea3c7e3f3d0bbab0734\"\n        ),\n        \"heads_sha256\": PINNED_HEADS_SHA256,\n        \"parent\": \"E2 captured 20-member DINOv2 rank ensemble\",\n        \"pixel_rules\": dict(RULES),\n    }\n    if not primary.is_file():\n        raise FileNotFoundError(\"E2 parent submission is absent\")\n    shutil.copy2(primary, preserved)\n\n    try:\n        device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\n        if device.type != \"cuda\":\n            raise RuntimeError(\"E9b RadImageNet inference requires CUDA\")\n        elapsed = max(0.0, time.time() - float(globals().get(\"T0\", time.time())))\n        available = 8.72 * 3600 - elapsed\n        audit[\"elapsed_before_e9b_seconds\"] = elapsed\n        audit[\"available_at_start_seconds\"] = available\n        if available < 45 * 60:\n            raise TimeoutError(f\"only {available / 60:.1f} minutes remain\")\n\n        pinned = _v52_load_pinned_e9b()\n        audit[\"oof_gate\"] = {\n            \"alpha\": pinned[\"alpha\"],\n            \"selected_targets\": pinned[\"selected_targets\"],\n            \"preserved_targets\": pinned[\"preserved_targets\"],\n            \"public_e2_macro_auc\": pinned[\"public_base\"],\n            \"public_selective_macro_auc\": pinned[\"public_selective\"],\n            \"remote_e2_macro_auc\": pinned[\"remote_base\"],\n            \"remote_selective_macro_auc\": pinned[\"remote_selective\"],\n            \"remote_outer_fold_choices\": pinned[\"remote_outer_fold_choices\"],\n            \"selection_supported\": True,\n        }\n\n        test = pd.read_csv(ROOT / \"test.csv\", dtype={\"StudyInstanceUID\": str})\n        test_series = pd.read_csv(\n            ROOT / \"test_series.csv\",\n            dtype={\"StudyInstanceUID\": str, \"SeriesInstanceUID\": str},\n        )\n        plane = dict(zip(test_series.SeriesInstanceUID, test_series.Anatomical_Plane))\n        headers = annotate(walk(\"test_series\"))\n        audit_official_sequence_metadata(headers, test_series)\n        studies, pixels, slot_mask = build_cache(\n            pick_slots(headers, plane), plane, lat_of(headers, \"test-e9b \"), \"test-e9b\"\n        )\n        by_uid = {str(uid): index for index, uid in enumerate(studies)}\n        missing = [uid for uid in test.StudyInstanceUID if uid not in by_uid]\n        if missing:\n            raise RuntimeError(f\"{len(missing)} test studies absent from E9b cache\")\n        order = np.asarray([by_uid[uid] for uid in test.StudyInstanceUID], dtype=np.int64)\n        pixels, slot_mask = pixels[order], slot_mask[order]\n        token_count = int(np.repeat(slot_mask[:, :, None], CACHE_SLICES, 2).sum())\n        if token_count < int(0.85 * len(test) * N_SLOT * CACHE_SLICES):\n            raise RuntimeError(f\"insufficient acquired E9b test slices: {token_count}\")\n        features, token_mask = encode_radimagenet(pixels, slot_mask, device)\n        del pixels, slot_mask, headers\n        gc.collect()\n\n        all_test = np.arange(len(test), dtype=np.int64)\n        fold_predictions = []\n        for record in pinned[\"folds\"]:\n            head = FoundationQueryHead().to(device)\n            head.load_state_dict(record[\"state_dict\"], strict=True)\n            fold_predictions.append(\n                predict_head(head, features, token_mask, all_test, device)\n            )\n            del head\n            torch.cuda.empty_cache()\n        if len(fold_predictions) != 5:\n            raise RuntimeError(\"E9b test inference did not use all five heads\")\n        rad_test = np.mean(np.stack(fold_predictions), axis=0)\n        if rad_test.shape != (len(test), len(TARGETS)) or not np.isfinite(rad_test).all():\n            raise RuntimeError(f\"invalid E9b prediction shape/value: {rad_test.shape}\")\n\n        baseline = pd.read_csv(preserved, dtype={\"StudyInstanceUID\": str})\n        _v52_validate_submission(baseline, test.StudyInstanceUID)\n        rad_frame = pd.DataFrame(rad_test, columns=TARGETS)\n        rad_frame.insert(0, \"StudyInstanceUID\", test.StudyInstanceUID)\n        _v52_validate_submission(rad_frame, test.StudyInstanceUID)\n        rad_frame.to_csv(output / \"submission_rad_only.csv\", index=False)\n        baseline_rank = _v52_rank_columns(baseline[TARGETS].to_numpy())\n        rad_rank = _v52_rank_columns(rad_test)\n        selected = baseline.copy()\n        alpha = pinned[\"alpha\"]\n        for target in pinned[\"selected_targets\"]:\n            index = TARGETS.index(target)\n            selected[target] = (\n                (1.0 - alpha) * baseline_rank[:, index] + alpha * rad_rank[:, index]\n            )\n        for target in pinned[\"preserved_targets\"]:\n            if not np.array_equal(\n                selected[target].to_numpy(), baseline[target].to_numpy()\n            ):\n                raise RuntimeError(f\"E9b failed to preserve {target}\")\n        _v52_validate_submission(selected, test.StudyInstanceUID)\n        selected_path = output / \"submission_e2_rad_robust_200.csv\"\n        selected.to_csv(selected_path, index=False)\n        selected = pd.read_csv(selected_path, dtype={\"StudyInstanceUID\": str})\n        _v52_validate_submission(selected, test.StudyInstanceUID)\n\n        audit.update({\n            \"test_studies\": len(test),\n            \"test_available_slice_tokens\": token_count,\n            \"test_head_count\": len(fold_predictions),\n            \"selected_path\": str(selected_path),\n            \"selected_sha256\": _v52_sha256(selected_path),\n            \"fallback_sha256\": _v52_sha256(preserved),\n        })\n        shutil.copy2(selected_path, primary)\n        if _v52_sha256(primary) != audit[\"selected_sha256\"]:\n            raise RuntimeError(\"primary E9b copy hash mismatch\")\n        audit[\"status\"] = \"CANDIDATE_SELECTED\"\n        log(\n            f\"E9b selected alpha={alpha:.3f} on \"\n            f\"{len(pinned['selected_targets'])} dual-OOF-stable targets; \"\n            \"Baker's and Fracture preserve E2\"\n        )\n    except Exception as error:\n        audit[\"status\"] = \"ERROR_E2_PRESERVED\"\n        audit[\"error\"] = f\"{type(error).__name__}: {error}\"\n        audit[\"traceback\"] = traceback.format_exc()\n        log(f\"E9b preserves E2: {audit['error']}\")\n    finally:\n        if audit.get(\"status\") != \"CANDIDATE_SELECTED\" and preserved.is_file():\n            shutil.copy2(preserved, primary)\n        audit[\"primary_sha256\"] = _v52_sha256(primary) if primary.is_file() else None\n        audit_path.write_text(json.dumps(audit, indent=2, sort_keys=True) + \"\\n\")\n\n\ndef _v52_rad_test_predictions(pinned, test, device, tag):\n    \"\"\"Five-head RadImageNet test prediction on the official test tree.\"\"\"\n    test_series = pd.read_csv(\n        ROOT / \"test_series.csv\",\n        dtype={\"StudyInstanceUID\": str, \"SeriesInstanceUID\": str},\n    )\n    plane = dict(zip(test_series.SeriesInstanceUID, test_series.Anatomical_Plane))\n    headers = annotate(walk(\"test_series\"))\n    audit_official_sequence_metadata(headers, test_series)\n    studies, pixels, slot_mask = build_cache(\n        pick_slots(headers, plane), plane, lat_of(headers, f\"{tag} \"), tag\n    )\n    by_uid = {str(uid): index for index, uid in enumerate(studies)}\n    missing = [uid for uid in test.StudyInstanceUID if uid not in by_uid]\n    if missing:\n        raise RuntimeError(f\"{len(missing)} test studies absent from {tag} cache\")\n    order = np.asarray([by_uid[uid] for uid in test.StudyInstanceUID], dtype=np.int64)\n    pixels, slot_mask = pixels[order], slot_mask[order]\n    token_count = int(np.repeat(slot_mask[:, :, None], CACHE_SLICES, 2).sum())\n    if token_count < int(0.85 * len(test) * N_SLOT * CACHE_SLICES):\n        raise RuntimeError(f\"insufficient acquired {tag} test slices: {token_count}\")\n    features, token_mask = encode_radimagenet(pixels, slot_mask, device)\n    del pixels, slot_mask, headers\n    gc.collect()\n\n    rows = np.arange(len(test), dtype=np.int64)\n    predictions = []\n    for record in pinned[\"folds\"]:\n        head = FoundationQueryHead().to(device)\n        head.load_state_dict(record[\"state_dict\"], strict=True)\n        predictions.append(predict_head(head, features, token_mask, rows, device))\n        del head\n        torch.cuda.empty_cache()\n    if len(predictions) != 5:\n        raise RuntimeError(f\"{tag} test inference did not use all five heads\")\n    rad_test = np.mean(np.stack(predictions), axis=0)\n    if rad_test.shape != (len(test), len(TARGETS)) or not np.isfinite(rad_test).all():\n        raise RuntimeError(f\"invalid {tag} prediction shape/value: {rad_test.shape}\")\n    return rad_test, token_count, len(predictions)\n\n\ndef main_v52_e10():\n    \"\"\"Deploy the audited E10 weight map; preserve the E2 parent on any failure.\"\"\"\n    import shutil\n\n    output = Path(\"/kaggle/working/rsna_rad_e10\")\n    output.mkdir(parents=True, exist_ok=True)\n    primary = Path(\"/kaggle/working/submission.csv\")\n    preserved = Path(\"/kaggle/working/submission_e2_preserved.csv\")\n    audit_path = Path(\"/kaggle/working/rad_e10_audit.json\")\n    audit = {\n        \"status\": \"E2_PRESERVED\",\n        \"mode\": \"pinned_v15_heads_inference_only\",\n        \"experiment\": \"E10\",\n        \"configuration\": E10_CONFIG,\n        \"evidence_boundary\": (\n            \"OOF values are 58-study diagnostics on official train labels, not Kaggle \"\n            \"scores. E10 widens the weight ladder that E9 truncated at 0.25 and votes only \"\n            \"where two independent OOF runs agree at the deployed weight.\"\n        ),\n        \"encoder\": \"RadImageNet ResNet-50 official PyTorch release\",\n        \"encoder_license\": \"CC-BY-NC-SA-4.0\",\n        \"heads_sha256\": PINNED_HEADS_SHA256,\n        \"contract_sha256\": PINNED_E10_CONTRACT_SHA256,\n        \"parent\": \"E2 captured 20-member DINOv2 rank ensemble\",\n        \"pixel_rules\": dict(RULES),\n    }\n    if not primary.is_file():\n        raise FileNotFoundError(\"E2 parent submission is absent\")\n    shutil.copy2(primary, preserved)\n\n    try:\n        device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\n        if device.type != \"cuda\":\n            raise RuntimeError(\"E10 RadImageNet inference requires CUDA\")\n        elapsed = max(0.0, time.time() - float(globals().get(\"T0\", time.time())))\n        available = 8.72 * 3600 - elapsed\n        audit[\"elapsed_before_e10_seconds\"] = elapsed\n        audit[\"available_at_start_seconds\"] = available\n        if available < 45 * 60:\n            raise TimeoutError(f\"only {available / 60:.1f} minutes remain\")\n\n        pinned = _v52_load_e10()\n        audit[\"weight_gate\"] = {\n            \"configuration\": pinned[\"configuration\"],\n            \"alpha_map\": pinned[\"alpha_map\"],\n            \"preserved_targets\": pinned[\"preserved_targets\"],\n            \"diagnostic_macro\": pinned[\"diagnostic_macro\"],\n            \"recomputed_macro\": pinned[\"recomputed_macro\"],\n            \"all_dual_positive\": pinned[\"all_dual_positive\"],\n            \"base_gold_macro_auc\": pinned[\"contract\"][\"base_gold_macro_auc\"],\n            \"rationale\": pinned[\"rationale\"],\n            \"remote_ladder_recomputed_in_kernel\": True,\n        }\n\n        test = pd.read_csv(ROOT / \"test.csv\", dtype={\"StudyInstanceUID\": str})\n        rad_test, token_count, head_count = _v52_rad_test_predictions(\n            pinned, test, device, \"test-e10\"\n        )\n\n        baseline = pd.read_csv(preserved, dtype={\"StudyInstanceUID\": str})\n        _v52_validate_submission(baseline, test.StudyInstanceUID)\n        rad_frame = pd.DataFrame(rad_test, columns=TARGETS)\n        rad_frame.insert(0, \"StudyInstanceUID\", test.StudyInstanceUID)\n        _v52_validate_submission(rad_frame, test.StudyInstanceUID)\n        rad_frame.to_csv(output / \"submission_rad_only.csv\", index=False)\n        baseline_rank = _v52_rank_columns(baseline[TARGETS].to_numpy())\n        rad_rank = _v52_rank_columns(rad_test)\n\n        # Materialise every audited rung so the ladder is inspectable from one run; only the\n        # configured map is promoted to the visible submission.\n        written = {}\n        for name, configuration in sorted(pinned[\"contract\"][\"configurations\"].items()):\n            frame = baseline.copy()\n            for target, alpha in configuration[\"alpha_map\"].items():\n                alpha = float(alpha)\n                if alpha > 0:\n                    index = TARGETS.index(target)\n                    frame[target] = (\n                        (1.0 - alpha) * baseline_rank[:, index] + alpha * rad_rank[:, index]\n                    )\n            for target, alpha in configuration[\"alpha_map\"].items():\n                if float(alpha) == 0.0 and not np.array_equal(\n                    frame[target].to_numpy(), baseline[target].to_numpy()\n                ):\n                    raise RuntimeError(f\"E10 failed to preserve {target} in {name}\")\n            _v52_validate_submission(frame, test.StudyInstanceUID)\n            path = output / f\"submission_e10_{name}.csv\"\n            frame.to_csv(path, index=False)\n            written[name] = _v52_sha256(path)\n        audit[\"ladder_sha256\"] = written\n\n        selected_path = output / f\"submission_e10_{pinned['configuration']}.csv\"\n        selected = pd.read_csv(selected_path, dtype={\"StudyInstanceUID\": str})\n        _v52_validate_submission(selected, test.StudyInstanceUID)\n        for target in pinned[\"preserved_targets\"]:\n            if not np.array_equal(\n                selected[target].to_numpy(), baseline[target].to_numpy()\n            ):\n                raise RuntimeError(f\"E10 promoted file does not preserve {target}\")\n        audit.update({\n            \"test_studies\": len(test),\n            \"test_available_slice_tokens\": token_count,\n            \"test_head_count\": head_count,\n            \"selected_path\": str(selected_path),\n            \"selected_sha256\": _v52_sha256(selected_path),\n            \"fallback_sha256\": _v52_sha256(preserved),\n        })\n        shutil.copy2(selected_path, primary)\n        if _v52_sha256(primary) != audit[\"selected_sha256\"]:\n            raise RuntimeError(\"primary E10 copy hash mismatch\")\n        audit[\"status\"] = \"CANDIDATE_SELECTED\"\n        voted = sorted(t for t, alpha in pinned[\"alpha_map\"].items() if alpha > 0)\n        log(\n            f\"E10 promoted {pinned['configuration']} over {len(voted)} dual-OOF targets; \"\n            f\"{', '.join(pinned['preserved_targets'])} preserve E2\"\n        )\n    except Exception as error:\n        audit[\"status\"] = \"ERROR_E2_PRESERVED\"\n        audit[\"error\"] = f\"{type(error).__name__}: {error}\"\n        audit[\"traceback\"] = traceback.format_exc()\n        log(f\"E10 preserves E2: {audit['error']}\")\n    finally:\n        if audit.get(\"status\") != \"CANDIDATE_SELECTED\" and preserved.is_file():\n            shutil.copy2(preserved, primary)\n        audit[\"primary_sha256\"] = _v52_sha256(primary) if primary.is_file() else None\n        audit_path.write_text(json.dumps(audit, indent=2, sort_keys=True) + \"\\n\")\n\n\ndef _v52_e11_availability(headers, plane_map):\n    \"\"\"Count studies offering each (plane, fat-suppression) pair before any slot is picked.\n\n    The parent arm reads only fat-suppressed series and never had to ask how many studies\n    carry a non-suppressed one. E11 depends on that answer, so it is measured and logged\n    rather than assumed: a slot nobody can fill is a masked column, and a run that produced\n    one silently would look like a weak arm instead of an absent input.\n    \"\"\"\n    frame = headers[[\"StudyInstanceUID\", \"SeriesInstanceUID\", \"fatsat\"]].copy()\n    frame[\"plane\"] = frame.SeriesInstanceUID.map(plane_map)\n    total = frame.StudyInstanceUID.nunique()\n    table = {}\n    for plane in (\"Sagittal\", \"Coronal\", \"Axial\"):\n        for fatsat in (True, False):\n            selected = frame[(frame.plane == plane) & (frame.fatsat == bool(fatsat))]\n            studies = selected.StudyInstanceUID.nunique()\n            key = f\"{plane}_{'FS' if fatsat else 'NOFS'}\"\n            table[key] = {\n                \"studies\": int(studies),\n                \"fraction\": float(studies / total) if total else 0.0,\n                \"series\": int(len(selected)),\n            }\n            log(f\"E11 availability {key}: {studies}/{total} studies, {len(selected)} series\")\n    return table\n\n\ndef main_v52_e11():\n    \"\"\"Train a third arm on a deliberately different pixel recipe. Never ships a candidate.\"\"\"\n    import shutil\n    from sklearn.model_selection import GroupKFold\n\n    output = Path(\"/kaggle/working/rsna_rad_e11\")\n    output.mkdir(parents=True, exist_ok=True)\n    primary = Path(\"/kaggle/working/submission.csv\")\n    preserved = Path(\"/kaggle/working/submission_e2_preserved.csv\")\n    audit_path = Path(\"/kaggle/working/rad_e11_audit.json\")\n    audit = {\n        \"status\": \"E2_PRESERVED\",\n        \"mode\": \"e11-diverse-recipe-training-only\",\n        \"evidence_boundary\": (\n            \"Every value here is a local out-of-fold diagnostic on 58 official image \"\n            \"labels. It is not a Kaggle score, and this mode never replaces the parent \"\n            \"submission under any outcome.\"\n        ),\n        \"recipe\": {\n            \"slots\": [list(slot) for slot in E11_SLOTS],\n            \"crop_mm\": E11_CROP_MM,\n            \"cache_slices\": E11_CACHE_SLICES,\n            \"img\": E11_IMG,\n            \"differs_from_parent_arm\": (\n                \"parent reads 3 fat-suppressed slots at full frame; this reads 3 \"\n                \"non-suppressed slots plus 1 suppressed anchor at a 130 mm physical crop\"\n            ),\n        },\n        \"encoder\": \"RadImageNet ResNet-50 official PyTorch release\",\n        \"encoder_license\": \"CC-BY-NC-SA-4.0 (Kaggle-hosted weight metadata)\",\n        \"encoder_sha256\": \"08629f7e7bd3e29b8ee9522ca3f65ce4d010a7ddf74f0ea3c7e3f3d0bbab0734\",\n    }\n    if not primary.is_file():\n        raise FileNotFoundError(\"E2 parent submission is absent\")\n    shutil.copy2(primary, preserved)\n\n    try:\n        device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\n        if device.type != \"cuda\":\n            raise RuntimeError(\"E11 training requires CUDA\")\n        elapsed = max(0.0, time.time() - float(globals().get(\"T0\", time.time())))\n        available = TIME_BUDGET - elapsed\n        audit[\"elapsed_before_e11_seconds\"] = elapsed\n        audit[\"available_at_start_seconds\"] = available\n        if available < 2.5 * 3600:\n            raise TimeoutError(f\"only {available / 60:.1f} minutes remain\")\n\n        # Point the shared pixel path at the E11 recipe. These are the same process globals\n        # the parent notebook's readers consult, so the override has to happen before any\n        # slot is picked or any pixel is decoded, and nothing after this point may assume\n        # the parent arm's values.\n        globals().update(\n            SLOTS=list(E11_SLOTS),\n            N_SLOT=len(E11_SLOTS),\n            CACHE_SLICES=int(E11_CACHE_SLICES),\n            IMG=int(E11_IMG),\n            CACHE_IMG=int(E11_IMG),\n            CROP_MM=float(E11_CROP_MM),\n        )\n        log(\n            f\"E11 recipe: {[s[0] for s in E11_SLOTS]} at {E11_CROP_MM:.0f} mm, \"\n            f\"{E11_IMG} px, {E11_CACHE_SLICES} slices/slot\"\n        )\n\n        train = pd.read_csv(ROOT / \"train.csv\", dtype={\"StudyInstanceUID\": str})\n        train_series = pd.read_csv(\n            ROOT / \"train_series.csv\",\n            dtype={\"StudyInstanceUID\": str, \"SeriesInstanceUID\": str},\n        )\n        if len(train) != 4407:\n            raise RuntimeError(f\"unexpected train study count {len(train)}\")\n        plane = dict(zip(train_series.SeriesInstanceUID, train_series.Anatomical_Plane))\n        headers = annotate(walk(\"train_series\"))\n        audit[\"availability\"] = _v52_e11_availability(headers, plane)\n\n        studies, pixels, slot_mask = build_cache(\n            pick_slots(headers, plane), plane, lat_of(headers, \"train-e11 \"), \"train-e11\"\n        )\n        by_uid = {str(uid): i for i, uid in enumerate(studies)}\n        missing = [uid for uid in train.StudyInstanceUID if uid not in by_uid]\n        if missing:\n            raise RuntimeError(f\"{len(missing)} train studies absent from cache\")\n        order = np.array([by_uid[uid] for uid in train.StudyInstanceUID], dtype=np.int64)\n        pixels, slot_mask = pixels[order], slot_mask[order]\n        fill = float(slot_mask.mean())\n        per_slot = {\n            name: float(slot_mask[:, k].mean())\n            for k, (name, _, _, _) in enumerate(E11_SLOTS)\n        }\n        audit[\"slot_fill\"] = per_slot\n        audit[\"overall_fill\"] = fill\n        for name, value in per_slot.items():\n            log(f\"E11 slot fill {name}: {value:.1%}\")\n        if fill < E11_MIN_FILL:\n            raise RuntimeError(f\"E11 slot fill {fill:.1%} below the {E11_MIN_FILL:.0%} floor\")\n\n        features, token_mask = encode_radimagenet(pixels, slot_mask, device)\n        del pixels, slot_mask, headers\n        gc.collect()\n\n        y, weights, gold = make_targets(train)\n        if int(gold.sum()) != 58:\n            raise RuntimeError(f\"expected 58 fully gold studies, observed {int(gold.sum())}\")\n        groups = report_groups(train)\n        if len(np.unique(groups)) < 4000:\n            raise RuntimeError(\"unexpected report-group collapse\")\n\n        splits = list(GroupKFold(5).split(features, groups=groups))\n        fold_id = np.full(len(train), -1, dtype=np.int8)\n        folds = []\n        oof = np.zeros_like(y, dtype=np.float32)\n        for fold, (tr, va) in enumerate(splits):\n            if set(groups[tr]).intersection(groups[va]):\n                raise RuntimeError(f\"report leakage in fold {fold}\")\n            fold_id[va] = fold\n            state, score = train_fold(\n                features, token_mask, y, weights, tr, va, fold, device\n            )\n            if state is None:\n                raise RuntimeError(f\"fold {fold} produced no checkpoint\")\n            head = FoundationQueryHead().to(device)\n            head.load_state_dict(state, strict=True)\n            oof[va] = predict_head(head, features, token_mask, va, device)\n            folds.append({\"fold\": fold, \"weak_auc\": float(score), \"state_dict\": state})\n            del head\n            torch.cuda.empty_cache()\n        if (fold_id < 0).any() or not np.isfinite(oof).all():\n            raise RuntimeError(\"incomplete E11 OOF\")\n\n        weak_auc = macro_auc(y, oof)\n        gold_auc = macro_auc(y[gold], oof[gold])\n        audit[\"weak_oof_auc\"] = float(weak_auc)\n        audit[\"gold_oof_auc\"] = float(gold_auc)\n        log(f\"E11 OOF weak macro AUC {weak_auc:.5f}\")\n        log(f\"E11 OOF gold macro AUC {gold_auc:.5f} on 58 studies\")\n\n        # The question E11 exists to answer is not whether this arm is strong on its own but\n        # whether it says something the portfolio does not already know. Both halves are\n        # measured against the same 58 rows and the same rank basis the deployed blend uses.\n        base_npz = find_input_file(\"oof.npz\")\n        with np.load(base_npz, allow_pickle=False) as bundle:\n            if bundle[\"targets\"].astype(str).tolist() != TARGETS:\n                raise RuntimeError(\"E11 E2 OOF target order drift\")\n            if not np.array_equal(\n                bundle[\"ids\"].astype(str), train.StudyInstanceUID.astype(str).to_numpy()\n            ):\n                raise RuntimeError(\"E11 E2 OOF study order drift\")\n            base_prediction = bundle[\"pred\"].astype(np.float64)\n        base = _v52_rank_columns(base_prediction[gold])\n        new = _v52_rank_columns(oof[gold].astype(np.float64))\n        gold_y = train.loc[gold, TARGETS].to_numpy(np.float64)\n        reference = _v52_target_auc(gold_y, base)\n        audit[\"e2_base_gold_macro\"] = float(np.mean([reference[t] for t in TARGETS]))\n        ladder = {}\n        for alpha in (0.20, 0.35, 0.50):\n            scores = _v52_target_auc(gold_y, (1.0 - alpha) * base + alpha * new)\n            ladder[f\"{alpha:.2f}\"] = {\n                \"macro\": float(np.mean([scores[t] for t in TARGETS])),\n                \"per_target_delta\": {t: float(scores[t] - reference[t]) for t in TARGETS},\n            }\n            log(f\"E11 blend alpha={alpha:.2f} gold macro {ladder[f'{alpha:.2f}']['macro']:.5f}\")\n        audit[\"blend_vs_e2\"] = ladder\n\n        try:\n            parent_oof = pd.read_csv(\n                find_input_file(\"v52_oof.csv\"), dtype={\"StudyInstanceUID\": str}\n            )\n            aligned = train[[\"StudyInstanceUID\"]].merge(\n                parent_oof, on=\"StudyInstanceUID\", how=\"left\", validate=\"one_to_one\"\n            )\n            parent = _v52_rank_columns(aligned[TARGETS].to_numpy(np.float64)[gold])\n            audit[\"correlation_with_parent_arm\"] = {\n                t: float(np.corrcoef(parent[:, i], new[:, i])[0, 1])\n                for i, t in enumerate(TARGETS)\n            }\n            log(\n                \"E11 mean rank correlation with the parent arm: \"\n                f\"{np.mean(list(audit['correlation_with_parent_arm'].values())):.3f}\"\n            )\n        except FileNotFoundError:\n            audit[\"correlation_with_parent_arm\"] = None\n\n        torch.save(\n            {\n                \"version\": \"e11-radimagenet-resnet50-diverse-1\",\n                \"targets\": TARGETS,\n                \"encoder_sha256\": audit[\"encoder_sha256\"],\n                \"slots\": [list(slot) for slot in E11_SLOTS],\n                \"crop_mm\": E11_CROP_MM,\n                \"img\": E11_IMG,\n                \"slices_per_plane\": E11_CACHE_SLICES,\n                \"feature\": \"global_average_pool\",\n                \"folds\": folds,\n                \"weak_oof_auc\": float(weak_auc),\n                \"gold_oof_auc\": float(gold_auc),\n            },\n            output / \"v52_e11_heads.pt\",\n        )\n        oof_frame = pd.DataFrame(oof, columns=TARGETS)\n        oof_frame.insert(0, \"StudyInstanceUID\", train.StudyInstanceUID)\n        oof_frame[\"fold\"] = fold_id\n        oof_frame[\"is_gold\"] = gold.astype(np.uint8)\n        oof_frame.to_csv(output / \"v52_e11_oof.csv\", index=False)\n        audit[\"status\"] = \"E11_TRAINED_E2_PRESERVED\"\n        audit[\"heads_sha256\"] = _v52_sha256(output / \"v52_e11_heads.pt\")\n        audit[\"oof_sha256\"] = _v52_sha256(output / \"v52_e11_oof.csv\")\n    except Exception as error:\n        audit[\"status\"] = \"ERROR_E2_PRESERVED\"\n        audit[\"error\"] = f\"{type(error).__name__}: {error}\"\n        audit[\"traceback\"] = traceback.format_exc()\n        log(f\"E11 preserves E2: {audit['error']}\")\n    finally:\n        if preserved.is_file():\n            shutil.copy2(preserved, primary)\n        audit[\"primary_sha256\"] = _v52_sha256(primary) if primary.is_file() else None\n        audit_path.write_text(json.dumps(audit, indent=2, sort_keys=True) + \"\\n\")\n\n\nif ARM_MODE == \"e11\":\n    main_v52_e11()\nelse:\n    try:\n        find_input_file(\"v52_radimagenet_heads.pt\")\n    except FileNotFoundError:\n        main_v52()\n    else:\n        try:\n            find_input_file(\"e10_contract.json\")\n        except FileNotFoundError:\n            main_v52_pinned_e9b()\n        else:\n            main_v52_e10()\n\n","metadata":{},"outputs":[],"execution_count":null}]}