{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.13"},"papermill":{"default_parameters":{},"duration":4265.609901,"end_time":"2026-08-07T05:18:53.721406+00:00","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-08-07T04:07:48.111505+00:00","version":"2.7.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"04e716b1df4d4cfab5e01e8ac158194b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_449e07b57058497e94dbfcf6ffea33da","placeholder":"​","style":"IPY_MODEL_270356f07b874bf7842443231d6fa0fc","tabbable":null,"tooltip":null,"value":"Loading weights: 100%"}},"270356f07b874bf7842443231d6fa0fc":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"3a1037c1b5fd448581c3ef09d7e0d51f":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_708dc323c8cd4b3faad890af05e98dba","max":223,"min":0,"orientation":"horizontal","style":"IPY_MODEL_9994b8d39d6b49988a8148c2ac8a6250","tabbable":null,"tooltip":null,"value":223}},"449e07b57058497e94dbfcf6ffea33da":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"5ff7b09eaf3540b4938d5983605405f1":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_04e716b1df4d4cfab5e01e8ac158194b","IPY_MODEL_3a1037c1b5fd448581c3ef09d7e0d51f","IPY_MODEL_a2b8396a99a64993a8e2b664c80b96cf"],"layout":"IPY_MODEL_a6dff7cfe84d4f518230b3ff17588995","tabbable":null,"tooltip":null}},"708dc323c8cd4b3faad890af05e98dba":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8be87c4aa90c474c996bc1d07b47daba":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"9994b8d39d6b49988a8148c2ac8a6250":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"a2b8396a99a64993a8e2b664c80b96cf":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_8be87c4aa90c474c996bc1d07b47daba","placeholder":"​","style":"IPY_MODEL_c2f1af3f3b6046488d3da5b78b9d08d7","tabbable":null,"tooltip":null,"value":" 223/223 [00:00&lt;00:00, 968.49it/s, Materializing param=layernorm.weight]"}},"a6dff7cfe84d4f518230b3ff17588995":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"c2f1af3f3b6046488d3da5b78b9d08d7":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\nimport os\n\n# ---- edit here -------------------------------------------------------------------- #\nos.environ.setdefault(\"RSNA_FOLDS\", \"1\")            # 1 to validate, 5 for the real run\nos.environ.setdefault(\"RSNA_TIME_BUDGET\", \"14400\")  # seconds; raise to 28800 for 5 folds\n# os.environ[\"RSNA_BACKBONE\"]  = \"resnet50\"         # a timm name, overrides DINOv2\n# os.environ[\"SLOT_SCHEME\"]    = \"public\"           # ablate the header recovery\n# ------------------------------------------------------------------------------------ #\n\nfor k in (\"RSNA_FOLDS\", \"RSNA_TIME_BUDGET\", \"RSNA_BACKBONE\", \"SLOT_SCHEME\"):\n    if k in os.environ:\n        print(f\"{k} = {os.environ[k]}\")","metadata":{"execution":{"iopub.status.busy":"2026-08-07T08:19:40.358708Z","iopub.execute_input":"2026-08-07T08:19:40.359174Z","iopub.status.idle":"2026-08-07T08:19:40.366337Z","shell.execute_reply.started":"2026-08-07T08:19:40.359137Z","shell.execute_reply":"2026-08-07T08:19:40.365118Z"},"papermill":{"duration":0.013698,"end_time":"2026-08-07T04:07:50.305827+00:00","exception":false,"start_time":"2026-08-07T04:07:50.292129+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"Report -> twelve graded targets, in nine languages.\n\nProvenance: the multilingual lexicon and the assert/negate/hedge scoring are adapted\nfrom the community notebook `rsna-knee-baseline-v1`. What is added here is §2 of this\nmodule - `distill`, which fits a text model to the *rule* scores out-of-fold and blends\nit back in.\n\nThat addition exists because of a failure mode the rule extractor cannot fix from\ninside. A rule that never fires does not raise; it emits a negative. A lexicon that is\nthick in English and thin in Greek therefore looks like a corpus where Greek patients\nhave fewer findings, and because language tracks the reporting site, that bias is\naligned with a scanner and a population rather than averaging out as noise.\n\nA character n-gram model fitted to the rule scores over the *whole* corpus sees the\nGreek phrasings that co-occur with rule-positive reports and scores them, without anyone\nhaving written them into the lexicon. Fitted out-of-fold and grouped on report text, it\ncannot simply memorise the rules it was trained on.\n\"\"\"\n\nfrom __future__ import annotations\n\nimport re\nimport unicodedata\n\nimport numpy as np\n\nTARGETS = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\",\n    \"Medial OA\", \"Lateral OA\", \"PF OA\", \"Effusion\",\n    \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\",\n]","metadata":{"execution":{"iopub.status.busy":"2026-08-07T08:19:40.367997Z","iopub.execute_input":"2026-08-07T08:19:40.368644Z","iopub.status.idle":"2026-08-07T08:19:40.389274Z","shell.execute_reply.started":"2026-08-07T08:19:40.368598Z","shell.execute_reply":"2026-08-07T08:19:40.388086Z"},"papermill":{"duration":0.010988,"end_time":"2026-08-07T04:07:50.329902+00:00","exception":false,"start_time":"2026-08-07T04:07:50.318914+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Turkish dotted/dotless i must be folded before casefolding, otherwise \"İZLENMEZ\"\n# and \"izlenmez\" diverge. ss and the Croatian/Serbian d-with-stroke likewise.\n_PRE = str.maketrans({\n    \"ı\": \"i\", \"İ\": \"i\", \"I\": \"i\", \"ß\": \"ss\", \"đ\": \"d\", \"Đ\": \"d\",\n    \"ø\": \"o\", \"Ø\": \"o\", \"æ\": \"ae\", \"Æ\": \"ae\",\n})\n\n\ndef normalize(text: str) -> str:\n    \"\"\"Fold case, diacritics and separators; keep Greek and Cyrillic letters.\n\n    NFKD decomposition strips Latin accents and Greek tonos alike (a -> a), which is what\n    we want: reports are inconsistent about accents. It also maps the MICRO SIGN U+00B5\n    to a real mu, which matters because most Greek reports here use the wrong codepoint.\n    \"\"\"\n    if not isinstance(text, str):\n        return \"\"\n    text = text.translate(_PRE).lower()\n    text = unicodedata.normalize(\"NFKD\", text)\n    text = \"\".join(ch for ch in text if not unicodedata.combining(ch))\n    text = text.replace(\"­\", \"\")               # soft hyphen\n    text = re.sub(r\"[_\\-/\\\\]+\", \" \", text)\n    text = re.sub(r\"[ \\t]+\", \" \", text)\n    return text\n\n\n_SENT_SPLIT = re.compile(r\"(?<=[.;!?])\\s+|\\n+\")\n\n\ndef clauses(text: str):\n    \"\"\"Split into clauses, then attach `header:` lines to the value that follows.\n\n    A report line reading `Fractures :` followed by `Aucune.` is one statement. Splitting\n    on punctuation alone separates the anatomy from its negation and flips the label.\n    \"\"\"\n    norm = normalize(text)\n    raw = [c.strip() for c in _SENT_SPLIT.split(norm) if c and c.strip()]\n\n    merged = []\n    for i, c in enumerate(raw):\n        # A fragment ending in a colon is a heading for the next fragment. Structured\n        # English reports write long ones - \"lateral compartment (meniscus, collateral\n        # ligament complex, cartilage):\" is eight words - so the cap is generous.\n        #\n        # The heading is emitted ONLY joined to its value, never also on its own. A bare\n        # `Fractures :` contains the anatomy and no polarity cue, so scoring it alone\n        # reads it as an assertion - and the very next line is `Aucune.` Structured\n        # reports are built out of exactly this shape, so keeping the bare heading turns\n        # every negated section header into a false positive. The joined clause is a\n        # superset of the heading's text, so nothing is lost by dropping it.\n        if c.endswith(\":\") and len(c.split()) <= 14 and i + 1 < len(raw):\n            merged.append(c + \" \" + raw[i + 1])\n            continue\n        merged.append(c)\n    # Comma-separated enumerations inside a long clause hide separate assertions.\n    out = []\n    for c in merged:\n        out.append(c)\n        if len(c.split()) > 25:\n            out.extend(p.strip() for p in c.split(\",\") if len(p.split()) > 2)\n    return out\n\n\ndef _rx(*alts: str) -> re.Pattern:\n    return re.compile(\"|\".join(alts))\n\n\nNEGATION = _rx(\n    # `none` and `nil` matter more than their frequency suggests: a structured report\n    # writes the anatomy as a heading and the finding as a one-word value beneath it,\n    # so `FRACTURE:` / `None.` is the entire statement and missing that one word flips\n    # the label on every section that was checked and found clear.\n    r\"\\bno\\b\", r\"\\bnot\\b\", r\"\\bnone\\b\", r\"\\bnil\\b\", r\"\\babsent\\b\",\n    r\"\\bwithout\\b\", r\"\\bnegative\\b\", r\"\\babsence\\b\",\n    r\"\\bno evidence\\b\", r\"\\bunremarkable\\b\", r\"\\bfree of\\b\",\n    r\"\\bsin\\b\", r\"\\bno hay\\b\", r\"\\bausencia\\b\", r\"\\bausentes?\\b\", r\"\\bningun[ao]?\\b\",\n    r\"\\bpas de\\b\", r\"\\bsans\\b\", r\"\\baucune?\\b\", r\"\\brien\\b\",\n    r\"\\bgeen\\b\", r\"\\bzonder\\b\", r\"\\bniet\\b\",\n    r\"\\bkeine?\\b\", r\"\\bohne\\b\", r\"\\bnicht\\b\",\n    r\"\\byok\\b\", r\"\\byoktur\\b\", r\"izlenmemekte\", r\"saptanmadi\", r\"\\bdegil\\b\",\n    r\"gozlenmemekte\", r\"mevcut degil\", r\"eslik etmiyor\", r\"\\bizlenmedi\\b\",\n    r\"\\bnema\\b\", r\"\\bbez\\b\", r\"\\bnisu\\b\", r\"\\bnije\\b\",\n    r\"\\bδεν\\b\", r\"\\bχωρις\\b\", r\"ουδεν\",\n    r\"\\bбез\\b\", r\"\\bне\\b\", r\"липсва\", r\"\\bняма\\b\",\n)\n\nNORMALITY = _rx(\n    r\"\\bnormal\", r\"\\bintact\\b\", r\"\\bpreserved\\b\", r\"\\bwithin normal limits\\b\",\n    r\"limites normales\", r\"\\bconservad\", r\"\\bintegr\", r\"\\bnormales\\b\",\n    r\"\\bdoga(l|ll)\\b\", r\"korunmus\", r\"\\bnormaldir\\b\", r\"olagan\",\n    r\"\\buredn\", r\"\\bocuvan\", r\"\\bodrzan\", r\"\\bintakt\",\n    r\"φυσιολογικ\", r\"ακεραι\",\n    r\"unauffallig\", r\"regelrecht\",\n    r\"нормал\", r\"запазен\", r\"съхранен\", r\"\\bбез особености\\b\",\n    r\"\\bgaaf\\b\", r\"\\bnormaal\\b\",\n)\n\nUNCERTAIN = _rx(\n    r\"\\bpossible\\b\", r\"\\bprobable\\b\", r\"\\bsuspicious\\b\", r\"\\bsuspected\\b\",\n    r\"cannot (be )?exclude\", r\"\\bmay\\b\", r\"\\bquestionable\\b\", r\"\\bequivocal\\b\",\n    r\"\\bposible\\b\", r\"sin criterios categoricos\", r\"\\bdudos\",\n    r\"\\bmuhtemel\\b\", r\"\\bolasi\\b\", r\"\\bsupheli\\b\", r\"\\bizlenim\",\n    r\"\\bmoguce\\b\", r\"\\bvjerojatno\\b\", r\"\\bsumnja\\b\",\n    r\"πιθαν\", r\"υποπτ\",\n    r\"\\bmoglich\", r\"\\bverdachtig\", r\"\\bfraglich\", r\"\\bv\\.a\\.\\b\",\n    r\"\\bвъзможно\\b\", r\"\\bвероятно\\b\", r\"суспект\",\n    r\"\\bmogelijk\\b\", r\"\\bverdacht\\b\",\n)\n\nTEAR = _rx(\n    # `disrupt`, not `\\bdisruption\\b`: \"the ACL is disrupted\" is the commonest English\n    # phrasing for a complete tear and the noun form misses it entirely.\n    r\"\\btear\", r\"\\btorn\\b\", r\"\\brupture\", r\"disrupt\", r\"discontinuit\",\n    r\"\\bavuls\", r\"\\blacerat\",\n    r\"\\brotura\\b\", r\"\\broturas\\b\", r\"\\bruptura\", r\"\\bdesgarro\", r\"\\broto\\b\",\n    r\"\\bdechirure\", r\"\\bdechire\",\n    r\"\\bscheur\", r\"\\bruptuur\", r\"gescheurd\",\n    # German compounds the pathology onto the anatomy - Innenmeniskusriss,\n    # Kreuzbandruptur, Meniskusabriss - so these stems must not carry a leading word\n    # boundary or the entire German subcorpus reads as negative.\n    r\"riss\\b\", r\"einriss\", r\"ruptur\", r\"zerreiss\", r\"\\blasion\",\n    r\"kontinuitatsunterbrechung\",\n    r\"\\byirtik\", r\"\\byirtig\", r\"\\bkopma\\b\", r\"butunluk kaybi\",\n    r\"\\bpuknuce\", r\"\\bprekid\\b\", r\"\\bpukotin\",\n    r\"ρηξη\", r\"ρηξις\", r\"ρηγμα\",\n    r\"руптура\", r\"разкъсв\", r\"разрив\", r\"скъсв\",\n)\n\nDEGEN = _rx(\n    r\"degenerat\", r\"\\bmucoid\\b\", r\"\\bmyxoid\\b\", r\"\\bfray\", r\"\\bfissur\",\n    r\"dejeneratif\", r\"\\bmukoid\\b\", r\"degenerativn\", r\"εκφυλιστ\", r\"дегенерат\",\n    r\"\\bmuco ?ide\\b\", r\"aufgefasert\",\n)\n\nINJURY = _rx(\n    r\"\\binjur\", r\"\\bsprain\", r\"\\blesion\", r\"\\blasion\", r\"\\bedema\\b\", r\"\\boedema\\b\",\n    r\"\\bodem\\b\", r\"\\bedem\\b\", r\"\\bοιδημα\", r\"\\bодем\", r\"\\bедем\", r\"\\bstrain\\b\",\n    r\"\\bhigh signal\\b\", r\"\\bsignal alteration\\b\", r\"\\bhiperintens\", r\"\\bhyperintens\",\n    r\"\\bthicken\", r\"\\bzadebljanje\\b\", r\"\\bverdikking\\b\", r\"\\bdistenzij\",\n    r\"\\blaksite\\b\", r\"\\blaxity\\b\", r\"\\bpartial\\b\", r\"\\bparcijaln\", r\"\\bparcial\",\n    r\"\\bpartiel\", r\"\\bpartiell\",\n)\n\nANAT = {\n    \"ACL\": _rx(\n        r\"anterior cruciate\", r\"\\bacl\\b\",\n        r\"cruzado anterior\", r\"\\blca\\b\",\n        r\"croise anterieur\",\n        r\"voorste kruisband\", r\"\\bvkb\\b\",\n        r\"vorderes kreuzband\", r\"vorderen kreuzband\", r\"vordere kreuzband\",\n        r\"on capraz\", r\"\\bocb\\b\",\n        r\"prednji krizni\", r\"prednjeg krizn\",\n        r\"προσθι[οα][^ ]* χιαστ\", r\"προσθιου χιαστου\", r\"χιαστο[^ ]* συνδεσμ\",\n        r\"предна кръстна\", r\"предната кръстна\",\n        # Plural, unqualified: reports routinely clear both cruciates in one clause\n        # (\"Ligamentos cruzados y colaterales dentro de limites normales\").\n        r\"cruciate ligaments\", r\"ligamentos cruzados\", r\"ligaments croises\",\n        r\"kruisbanden\", r\"kreuzbander\", r\"capraz baglar\", r\"krizn[a-z]* ligament[a-z]*\",\n        r\"χιαστοι συνδεσμ\", r\"χιαστων συνδεσμ\", r\"кръстните връзки\", r\"кръстни връзки\",\n    ),\n    \"MCL\": _rx(\n        r\"medial collateral\", r\"\\bmcl\\b\", r\"tibial collateral\",\n        r\"colateral medial\", r\"colateral interno\", r\"\\blcm\\b\",\n        r\"collateral medial\", r\"collateral interne\",\n        r\"mediale collaterale\", r\"binnenband\",\n        r\"innenband\", r\"mediales? kollateral\",\n        r\"\\bic yan bag\", r\"medial kollateral\", r\"\\biyb\\b\",\n        r\"medijalni kolateraln\", r\"medijalnog kolateraln\",\n        r\"εσω πλαγι\", r\"εσωτερικο πλαγι\",\n        r\"медиален колатерал\", r\"вътрешна странична\",\n        # \"Ligamentos cruzados y colaterales\" separates the noun from its adjective, so\n        # the adjective has to stand alone as a cue.\n        r\"\\bcolaterales\\b\", r\"\\bcollateraux\\b\", r\"\\bcollateralen\\b\", r\"\\bkolateralni\\b\",\n        r\"collateral ligaments\", r\"ligamentos colaterales\", r\"ligaments collateraux\",\n        r\"collaterale banden\", r\"kollateralbander\", r\"seitenbander\", r\"yan baglar\",\n        r\"kolateraln[a-z]* ligament[a-z]*\", r\"πλαγιοι συνδεσμ\", r\"πλαγιων συνδεσμ\",\n        r\"колатерални връзки\", r\"страничните връзки\",\n    ),\n    \"Medial Meniscus\": _rx(\n        r\"medial meniscus\", r\"\\bmm\\b(?= tear)\", r\"medial menisc\",\n        r\"menisco medial\", r\"menisco interno\",\n        r\"menisque medial\", r\"menisque interne\",\n        r\"mediale meniscus\", r\"binnenmeniscus\",\n        r\"innenmeniskus\", r\"medialen? meniskus\", r\"innenmeniskushinterhorn\",\n        r\"medyal menisk\", r\"\\bic menisk\",\n        r\"medijalni meniskus\", r\"medijalnog meniskusa\", r\"medijalnom meniskusu\",\n        r\"εσω μηνισκ\", r\"μηνισκ[^ ]* του εσω\", r\"εσω διαμερισμα[^.]{0,40}μηνισκ\",\n        r\"медиалния менискус\", r\"медиален менискус\", r\"вътрешния менискус\",\n    ),\n    \"Lateral Meniscus\": _rx(\n        r\"lateral meniscus\", r\"lateral menisc\",\n        r\"menisco lateral\", r\"menisco externo\",\n        r\"menisque lateral\", r\"menisque externe\",\n        r\"laterale meniscus\", r\"buitenmeniscus\",\n        r\"aussenmeniskus\", r\"lateralen? meniskus\",\n        r\"lateral menisk\", r\"\\bdis menisk\",\n        r\"lateralni meniskus\", r\"lateralnog meniskusa\", r\"lateralnom meniskusu\",\n        r\"εξω μηνισκ\", r\"μηνισκ[^ ]* του εξω\", r\"εξω διαμερισμα[^.]{0,40}μηνισκ\",\n        r\"латералния менискус\", r\"латерален менискус\", r\"външния менискус\",\n    ),\n}\n\n# Osteoarthritis is rarely written as \"osteoarthritis\". It is written as cartilage loss,\n# chondropathy grade, joint space narrowing, or osteophytes - scoped to a compartment.\nOA_EVIDENCE = _rx(\n    r\"osteoarthrit\", r\"\\barthros\", r\"\\bgonarthros\", r\"\\bosteoarthros\",\n    r\"chondropath\", r\"chondromalac\", r\"condropat\", r\"condromalac\",\n    r\"cartilage loss\", r\"cartilage thinning\", r\"chondral (loss|defect|ulcer|thinning)\",\n    r\"osteophyt\", r\"osteofit\", r\"osteofyt\", r\"osteofito\", r\"osteophyten\",\n    r\"joint space narrowing\", r\"pinzamiento articular\",\n    r\"kikirdak kayb\", r\"kikirdak incelme\", r\"kondropati\", r\"kondral\",\n    r\"kraakbeen(lijden|verlies)\", r\"gonartrose\", r\"artrose\",\n    r\"knorpel(verlust|schaden|defekt)\", r\"arthrose\", r\"gonarthrose\",\n    r\"hrskavic\", r\"hondromalac\", r\"artroz\", r\"osteoartrit\",\n    r\"χονδρ[^ ]*παθ\", r\"αρθριτ\", r\"αρθρωσ\", r\"οστεοφυτ\",\n    r\"αρθρικου χονδρου\", r\"εξαλειψη του αρθρικου χονδρου\",\n    r\"артроз\", r\"хондропат\", r\"остеофит\", r\"хрущял[^.]{0,30}(изтън|увред|дефект)\",\n    r\"ulcera[s]? condral\", r\"cartilago[^.]{0,25}(perdida|adelgaz)\",\n    r\"icrs grade\", r\"outerbridge\",\n)\n\nCOMPARTMENT = {\n    \"Medial OA\": _rx(\n        r\"medial (femorotibial|tibiofemoral|compartment)\",\n        r\"compartimento femorotibial medial\", r\"femorotibial interno\",\n        r\"mediaal femorotibiaal\", r\"mediale femorotibial\",\n        r\"medial femorotibial\", r\"medialen kompartiment\", r\"innere[sn]? kompartiment\",\n        r\"medyal femorotibial\", r\"ic kompartman\", r\"medyal kompartman\",\n        r\"medijaln[^ ]* (femorotibi|odjelj|kompartm)\",\n        r\"εσω διαμερισμα\", r\"εσω κνημιαι\", r\"εσω μηριαι\",\n        r\"медиалн[^ ]* (компартм|отдел|тибиал|феморотиб)\",\n        r\"medial (femoral|tibial) (condyle|plateau)\", r\"condilo femoral medial\",\n        r\"medialen? (femurkondyl|tibiaplateau)\", r\"mediale femorale condyl\",\n    ),\n    \"Lateral OA\": _rx(\n        r\"lateral (femorotibial|tibiofemoral|compartment)\",\n        r\"compartimento femorotibial lateral\", r\"femorotibial externo\",\n        r\"lateraal femorotibiaal\", r\"laterale femorotibial\",\n        r\"lateral femorotibial\", r\"lateralen kompartiment\", r\"aussere[sn]? kompartiment\",\n        r\"dis kompartman\", r\"lateral kompartman\",\n        r\"lateraln[^ ]* (femorotibi|odjelj|kompartm)\",\n        r\"εξω διαμερισμα\", r\"εξω κνημιαι\", r\"εξω μηριαι\",\n        r\"латералн[^ ]* (компартм|отдел|тибиал|феморотиб)\",\n        r\"lateral (femoral|tibial) (condyle|plateau)\", r\"condilo femoral lateral\",\n        r\"lateralen? (femurkondyl|tibiaplateau)\", r\"laterale femorale condyl\",\n    ),\n    \"PF OA\": _rx(\n        r\"patellofemoral\", r\"femoropatellar\", r\"femoropatelar\", r\"patelofemoral\",\n        r\"retropatellar\", r\"retrorotulian\", r\"\\btrochlea\", r\"\\btroclea\", r\"\\btroklea\",\n        r\"\\bpatella\\b\", r\"\\bpatellar\\b\", r\"\\brotulian\", r\"\\brotula\\b\", r\"\\bpatele\\b\",\n        r\"\\bpatellae?\\b\", r\"patellofemoraal\", r\"femoropatellair\",\n        r\"επιγονατιδ\", r\"μηροεπιγονατιδ\", r\"τροχιλ\",\n        r\"пател\", r\"феморопател\", r\"тролх\",\n        r\"anterior compartment\", r\"compartimento anterior\", r\"prednj[^ ]* odjeljk\",\n    ),\n}\n\n# Self-declaring findings: the term itself is the finding.\nDIRECT = {\n    \"Effusion\": _rx(\n        r\"\\beffusion\", r\"joint fluid\", r\"intra ?articular fluid\", r\"\\bhydrops\\b\",\n        r\"derrame articular\", r\"\\bderrame\\b\", r\"liquido articular\",\n        r\"epanchement\",\n        r\"gewrichtsvocht\", r\"\\bvocht\\b\", r\"gewrichtseffusie\",\n        r\"gelenkerguss\", r\"\\berguss\\b\", r\"gelenksergu\",\n        # \"diz eklemi ici sivi miktari ... artmis\" - the noun takes a possessive suffix,\n        # so `eklem ` alone misses. Match the stem plus any suffix.\n        r\"eklem\\w* ic\\w* sivi\", r\"efuzyon\", r\"eklem sivisi\",\n        r\"sivi (miktari|artisi|birikimi)\", r\"sivi artis\", r\"\\bsivi\\b[^.]{0,25}artmis\",\n        r\"\\bizljev\", r\"\\bizliv\", r\"zglobn[^ ]* tekucin\", r\"\\bhidrops\\b\",\n        r\"αρθρικ[^ ]* υγρ\", r\"υγρου ενδαρθρικα\", r\"ενδαρθρικ[^ ]* υγρ\", r\"ποσοτητα υγρου\",\n        r\"ενδαρθρικ\", r\"αρθρικη συλλογη\", r\"υγρο στην αρθρωση\", r\"υγρου στην αρθρωση\",\n        r\"ставен излив\", r\"излив\", r\"ставна течност\", r\"синовиална течност\",\n    ),\n    \"Synovitis\": _rx(\n        r\"synovit\", r\"sinovit\", r\"synovial (thickening|proliferation|hypertroph)\",\n        r\"synoviale? (verdikking|proliferatie)\",\n        r\"synovialitis\", r\"synovialis(verdickung|proliferation)\",\n        r\"sinovijalitis\", r\"zadebljanje sinovij\",\n        r\"υμενιτιδα\", r\"συνοβιτιδα\", r\"υμενικ[^ ]* υπερτροφ\", r\"αρθρικου υμεν\",\n        r\"синовит\", r\"синовиал[^ ]* (задебел|пролифер)\",\n        r\"verdikkingen van (het )?synovium\", r\"pannus\",\n    ),\n    \"Baker's\": _rx(\n        r\"baker\", r\"popliteal cyst\", r\"quiste popliteo\", r\"quistes popliteos\",\n        r\"kyste poplite\", r\"popliteale? cyst\", r\"poplitealzyste\", r\"bakerzyste\",\n        r\"popliteal kist\", r\"\\bbakerova\\b\", r\"poplitealn[^ ]* cist\",\n        r\"κυστη baker\", r\"πολυχωρη συνοβιακη κυστη\", r\"κυστη του baker\",\n        r\"киста на бейкър\", r\"бейкърова киста\", r\"поплитеална киста\",\n        r\"gastrocnemio ?semimembranos\", r\"gastrocnemius semimembranosus burs\",\n    ),\n    \"Contusion\": _rx(\n        r\"\\bcontusion\", r\"bone bruise\", r\"bone marrow (o?edema|contusion)\",\n        r\"\\bkontuz\", r\"medular bone o?edema\", r\"marrow o?edema\",\n        r\"edema oseo\", r\"edema de medula osea\",\n        r\"oedeme osseux\",\n        r\"botcontusie\", r\"botoedeem\", r\"beenmergoedeem\", r\"botmergoedeem\",\n        r\"knochenmarkodem\", r\"knochenodem\", r\"kontusion\",\n        r\"kemik kontuzyonu\", r\"kemik iligi odemi\", r\"kemik odemi\",\n        r\"kostani edem\", r\"edem kosti\", r\"kontuzij\",\n        r\"οστεομυελικ[^ ]* οιδημα\", r\"οστικο οιδημα\", r\"μυελικο οιδημα\",\n        r\"костномозъчен едем\", r\"костен едем\", r\"контузионен\",\n    ),\n    \"Fracture\": _rx(\n        r\"\\bfractur\", r\"\\bfract\\b\",\n        r\"\\bfractura\", r\"\\bfracturas\\b\",\n        r\"\\bfractuur\", r\"\\bbreuk\\b\",\n        r\"\\bfraktur\", r\"\\bbruch\\b\",\n        r\"\\bkirik\\b\", r\"\\bkirigi\\b\",\n        r\"\\bprijelom\", r\"impresijsk[^ ]* fraktur\",\n        r\"καταγμα\", r\"καταγματ\",\n        r\"фрактур\", r\"счупван\", r\"фисур\",\n        r\"insufficiency fracture\", r\"stress fracture\", r\"avulsion fracture\",\n        r\"subchondral fracture\", r\"subkondral kiri\",\n    ),\n}\n\n# Terms that look like a finding but are not the finding being scored.\nDECOY = {\n    \"Fracture\": _rx(r\"no fracture\", r\"microfractur\", r\"\\bfracture (risk|prophyla)\"),\n    \"Baker's\": _rx(r\"meniscal cyst\", r\"quiste meniscal\", r\"ganglion\"),\n}\n\nPAIRED = {\"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\"}\nOA_TARGETS = {\"Medial OA\", \"Lateral OA\", \"PF OA\"}\n\nSTEM_MENISCUS = _rx(r\"menisc\\w*\", r\"menisk\\w*\", r\"μηνισκ\\w*\", r\"мениск\\w*\")\nSTEM_CRUCIATE = _rx(r\"cruciate\", r\"cruzado\", r\"croise\", r\"kruisband\", r\"kreuzband\",\n                    r\"capraz bag\\w*\", r\"krizn\\w*\", r\"χιαστ\\w*\", r\"кръстн\\w*\",\n                    r\"\\bacl\\b\", r\"\\bpcl\\b\", r\"\\blca\\b\", r\"\\blcp\\b\", r\"\\bvkb\\b\",\n                    r\"\\bhkb\\b\", r\"\\bocb\\b\", r\"\\bacb\\b\")\nSTEM_COLLATERAL = _rx(r\"collateral\\w*\", r\"colateral\\w*\", r\"kollateral\\w*\",\n                      r\"collaterale\\w*\", r\"kolateraln\\w*\", r\"yan bag\\w*\",\n                      r\"πλαγι\\w*\", r\"колатерал\\w*\", r\"странич\\w*\",\n                      r\"innenband\\w*\", r\"aussenband\\w*\", r\"binnenband\\w*\",\n                      r\"\\bmcl\\b\", r\"\\blcl\\b\", r\"\\blcm\\b\", r\"\\biyb\\b\")\n\nSIDE_MEDIAL = _rx(r\"\\bmedial\\w*\", r\"\\bmedyal\\w*\", r\"\\bmedijaln\\w*\", r\"\\bmediaal\\w*\",\n                  r\"\\bmediale\\w*\", r\"\\binterno\\w*\", r\"\\binterne\\w*\", r\"\\binnen\\w*\",\n                  r\"\\bic\\b\", r\"\\bunutarnj\\w*\", r\"\\bεσω\\w*\", r\"\\bεσωτερικ\\w*\",\n                  r\"\\bмедиал\\w*\", r\"\\bвътреш\\w*\", r\"\\btibial collateral\\b\")\nSIDE_LATERAL = _rx(r\"\\blateral\\w*\", r\"\\bexterno\\w*\", r\"\\bexterne\\w*\", r\"\\bdis\\b\",\n                   r\"\\blateraln\\w*\", r\"\\baussen\\w*\", r\"\\bbuiten\\w*\", r\"\\bεξω\\w*\",\n                   r\"\\bεξωτερικ\\w*\", r\"\\bлатерал\\w*\", r\"\\bвъншн\\w*\",\n                   r\"\\bfibular collateral\\b\", r\"\\bvanjsk\\w*\")\nSIDE_ANTERIOR = _rx(r\"\\banterior\\w*\", r\"\\bant\\b\", r\"\\bon\\b\", r\"\\bprednj\\w*\",\n                    r\"\\bvorder\\w*\", r\"\\bvoorste\\b\", r\"\\bπροσθι\\w*\", r\"\\bпредн\\w*\",\n                    r\"\\banteriyor\\w*\", r\"\\bavant\\b\", r\"\\banterieur\\w*\")\n\n# Fracture is the target whose stem varies most across the corpus.\nSTEM_FRACTURE = _rx(r\"fractur\\w*\", r\"fraktur\\w*\", r\"fractuur\\w*\", r\"\\bfract\\b\",\n                    r\"kiri[kg]\\w*\", r\"prijelom\\w*\", r\"lom kosti\", r\"\\bbreuk\\w*\",\n                    r\"\\bbruch\\w*\", r\"καταγμα\\w*\", r\"καταγματ\\w*\", r\"фрактур\\w*\",\n                    # NOT a bare `fissur\\w*`: \"fisuras condrales\" and \"full thickness\n                    # fissures in the articular cartilage\" describe cartilage, not bone.\n                    r\"счупван\\w*\", r\"fisur\\w* (osea|oseas|kost)\", r\"fissur\\w* kost\")\n\nSTEM_OA_COMPARTMENT = _rx(r\"compartment\\w*\", r\"compartimento\\w*\", r\"compartiment\\w*\",\n                          r\"kompartman\\w*\", r\"kompartiment\\w*\", r\"odjelj\\w*\",\n                          r\"διαμερισμα\\w*\", r\"компартм\\w*\", r\"\\bотдел\\w*\",\n                          r\"femorotibial\\w*\", r\"femorotibiaal\\w*\", r\"tibiofemoral\\w*\",\n                          r\"femoro tibial\\w*\", r\"κνημιαι\\w*\", r\"μηριαι\\w*\",\n                          r\"femoral condyl\\w*\", r\"tibial plateau\\w*\",\n                          r\"condilo femoral\", r\"platillo tibial\", r\"tibiaplateau\\w*\",\n                          r\"femurkondyl\\w*\", r\"femoralne? kondil\\w*\",\n                          r\"tibijaln\\w* plato\", r\"femoral kondil\\w*\",\n                          r\"tibia plato\", r\"tibyal plato\")\n\n\ndef _near(clause: str, stem_rx: re.Pattern, qual_rx: re.Pattern, window: int = 55):\n    \"\"\"True if a stem match has a qualifier within `window` characters either side.\n\n    Character windows rather than token windows, because word order differs: English\n    puts the side before the noun, Greek and Bulgarian often after, and Turkish\n    attaches it as a separate preceding adjective.\n    \"\"\"\n    for m in stem_rx.finditer(clause):\n        lo = max(0, m.start() - window)\n        hi = min(len(clause), m.end() + window)\n        if qual_rx.search(clause[lo:hi]):\n            return True\n    return False\n\n\nSTEM_RULES = {\n    \"ACL\": (STEM_CRUCIATE, SIDE_ANTERIOR),\n    \"MCL\": (STEM_COLLATERAL, SIDE_MEDIAL),\n    \"Medial Meniscus\": (STEM_MENISCUS, SIDE_MEDIAL),\n    \"Lateral Meniscus\": (STEM_MENISCUS, SIDE_LATERAL),\n    \"Medial OA\": (STEM_OA_COMPARTMENT, SIDE_MEDIAL),\n    \"Lateral OA\": (STEM_OA_COMPARTMENT, SIDE_LATERAL),\n}\n\nSEV_LOW = _rx(\n    r\"\\bsmall\\b\", r\"\\bminimal\\b\", r\"\\btrace\\b\", r\"\\bmild\\b\", r\"\\bslight\\b\",\n    r\"\\btiny\\b\", r\"\\bscant\\b\", r\"\\bmimimal\\b\", r\"\\bdiscrete\\b\", r\"\\bfocal\\b\",\n    r\"\\bleve\\b\", r\"\\bminim\", r\"\\bpeque\", r\"\\bligero\\b\", r\"\\bescaso\\b\", r\"\\bdiscreto\\b\",\n    r\"\\bhafif\\b\", r\"\\baz miktarda\\b\", r\"\\bsilik\\b\",\n    r\"\\bmanja\\b\", r\"\\bmanji\\b\", r\"\\bblago\\b\", r\"\\bdiskretn\", r\"\\bmalo\\b\",\n    r\"\\bgering\", r\"\\bdiskret\", r\"\\bkleine?r?\\b\", r\"\\bwenig\\b\", r\"\\bzarte?\\b\",\n    r\"\\bbeperkte?\\b\", r\"\\bgeringe\\b\", r\"\\bweinig\\b\", r\"\\blichte?\\b\",\n    r\"\\bηπι\", r\"\\bμικρ\", r\"\\bελαχιστ\",\n    r\"\\bминимал\", r\"\\bлек\", r\"\\bмалк\", r\"\\bнеголям\",\n)\n\nSEV_HIGH = _rx(\n    r\"\\blarge\\b\", r\"\\bmarked\\b\", r\"\\bmassive\\b\", r\"\\bsevere\\b\", r\"\\bextensive\\b\",\n    r\"\\bmoderate\\b\", r\"\\bgross\\b\", r\"\\bsignificant\\b\", r\"\\babundant\\b\", r\"\\btense\\b\",\n    r\"\\bmoderad\", r\"\\bimportante\\b\", r\"\\bsevera?\\b\", r\"\\bmarcad\", r\"\\bcuantios\",\n    r\"\\bbelirgin\\b\", r\"\\byaygin\\b\", r\"\\bileri\\b\", r\"\\bciddi\\b\", r\"\\bbol\\b\",\n    r\"\\bopsezan\\b\", r\"\\bveliki\\b\", r\"\\bizrazit\", r\"\\bznacajn\", r\"\\bumjeren\",\n    r\"\\bausgepragt\", r\"\\bdeutlich\", r\"\\bmassiv\", r\"\\bmassig\", r\"\\bgross\",\n    r\"\\buitgebreid\", r\"\\bgevorderd\", r\"\\bveel\\b\", r\"\\bmatige?\\b\",\n    r\"\\bμετρι\", r\"\\bμεγαλ\", r\"\\bεκτεταμεν\", r\"\\bευμεγεθ\", r\"\\bσοβαρ\",\n    r\"\\bголям\", r\"\\bизразен\", r\"\\bзначим\", r\"\\bумерен\", r\"\\bобилен\",\n)\n\n# OA is often asserted for the whole joint rather than per compartment\n# (\"tricompartmental osteoarthritis\", \"gonarthrose\"). Those statements are evidence for\n# all three OA targets.\nGLOBAL_OA = _rx(\n    r\"tri ?compartment\", r\"all three compartment\", r\"global(ised)? (oa|osteoarthrit)\",\n    r\"\\bgonarthros\", r\"\\bgonartros\", r\"\\bgonarthrose\", r\"\\bgonartrose\",\n    r\"osteoarthritis of the knee\", r\"artrosis (de |)(la )?rodilla\", r\"knee osteoarthrit\",\n    r\"\\bdiz osteoartrit\", r\"\\bgonartroz\", r\"artroza koljena\",\n    r\"οστεοαρθριτιδα\", r\"αρθριτιδα του γονατος\",\n    r\"артроза на колянната\", r\"гонартроз\",\n    r\"degenerative joint disease\", r\"\\bdjd\\b\",\n)\n\n# A bare \"bone marrow oedema\" is not a contusion when it sits under a cartilage defect:\n# subchondral oedema beneath a worn compartment is reactive degenerative signal, and\n# reading it as a bruise turns every osteoarthritic knee into a trauma case.\nDEGENERATIVE_MARROW = _rx(\n    r\"subchondral\", r\"subcondral\", r\"subkondral\", r\"supkondraln\", r\"subchondraln\",\n    r\"υποχονδρι\", r\"субхондрал\",\n    r\"\\bcyst\", r\"\\bquist\", r\"\\bzyste\\b\", r\"\\bcistic\", r\"reactive\", r\"reactivo\",\n)\n\nTRAUMA = _rx(\n    r\"\\bbruise\\b\", r\"\\bcontusion\", r\"\\bkontuz\",\n    r\"\\btrauma\", r\"\\bimpaction\\b\", r\"\\bpivot shift\\b\", r\"\\bkissing\\b\",\n    r\"\\bacute\\b\", r\"\\bagudo\\b\", r\"\\bakut\", r\"\\bpivot kaymasi\\b\",\n    r\"\\bbone bruise\\b\", r\"\\bbotcontusie\\b\",\n    r\"\\bконтузион\", r\"\\bμωλωπ\", r\"\\bkontuzij\",\n)\n\n\ndef _polarity(clause: str) -> str:\n    \"\"\"Classify one clause as positive, negative or uncertain for a matched term.\n\n    Scope is the whole clause. Clause segmentation already keeps statements short, and\n    a window in characters mis-scopes badly across languages with different word orders -\n    Turkish puts its negator at the end of the sentence, English at the front.\n    \"\"\"\n    if UNCERTAIN.search(clause):\n        return \"uncertain\"\n    if NEGATION.search(clause):\n        return \"negative\"\n    if NORMALITY.search(clause):\n        # \"meniscus normal\" negates; \"normal ... but tear\" does not.\n        if TEAR.search(clause) or re.search(r\"\\bgrade [34]\\b\", clause):\n            return \"positive\"\n        return \"negative\"\n    return \"positive\"\n\n\nclass _Matcher:\n    \"\"\"Phrase lexicon first, stem+side proximity as the fallback.\n\n    Exposes `.search` so it drops into the same slot as a compiled pattern.\n    \"\"\"\n\n    def __init__(self, phrase_rx, stem=None, side=None, window=55):\n        self.phrase_rx = phrase_rx\n        self.stem = stem\n        self.side = side\n        self.window = window\n\n    def search(self, clause):\n        m = self.phrase_rx.search(clause)\n        if m is not None:\n            return m\n        if self.stem is not None and _near(clause, self.stem, self.side, self.window):\n            return self.stem.search(clause)\n        return None\n\n\nANAT_MATCH = {tgt: _Matcher(ANAT[tgt], *STEM_RULES[tgt]) for tgt in PAIRED}\nCOMPARTMENT_MATCH = {\n    \"Medial OA\": _Matcher(COMPARTMENT[\"Medial OA\"], *STEM_RULES[\"Medial OA\"]),\n    \"Lateral OA\": _Matcher(COMPARTMENT[\"Lateral OA\"], *STEM_RULES[\"Lateral OA\"]),\n    \"PF OA\": _Matcher(COMPARTMENT[\"PF OA\"]),\n}\nDIRECT_MATCH = {\n    tgt: _Matcher(_rx(rx.pattern, STEM_FRACTURE.pattern) if tgt == \"Fracture\" else rx)\n    for tgt, rx in DIRECT.items()\n}\n\n\ndef _severity(clause: str) -> float:\n    \"\"\"Weight one positive mention by how emphatic the sentence is.\n\n    Ordered, not calibrated. A \"moderate effusion\" must outrank a \"trace effusion\" and\n    both must outrank silence; the absolute numbers do not matter to AUC.\n    \"\"\"\n    high = SEV_HIGH.search(clause) is not None\n    low = SEV_LOW.search(clause) is not None\n    if high and not low:\n        return 1.0\n    if low and not high:\n        return 0.45\n    return 0.75                       # unqualified mention\n\n\ndef _score_clauses(cls, anat_rx, path_rx=None, decoy_rx=None, context_penalty=None,\n                   context_bonus=None):\n    \"\"\"Accumulate graded evidence over clauses for one target.\n\n    Returns (score, confidence, n_pos, n_neg). Positives are graded by severity and by\n    optional context regexes; negatives only matter when nothing positive was found,\n    because reports assert normality for every structure they check.\n    \"\"\"\n    n_pos = n_neg = n_unc = 0\n    best = 0.0\n    for c in cls:\n        m = anat_rx.search(c)\n        if not m:\n            continue\n        if decoy_rx is not None and decoy_rx.search(c):\n            continue\n        if path_rx is not None and not path_rx.search(c):\n            if NORMALITY.search(c) and not NEGATION.search(c):\n                n_neg += 1\n            continue\n        pol = _polarity(c)\n        if pol == \"positive\":\n            n_pos += 1\n            w = _severity(c)\n            if context_penalty is not None and context_penalty.search(c):\n                w *= 0.45\n            if context_bonus is not None and context_bonus.search(c):\n                w = min(1.0, w * 1.35)\n            best = max(best, w)\n        elif pol == \"negative\":\n            n_neg += 1\n        else:\n            n_unc += 1\n            best = max(best, 0.30)\n\n    if n_pos or n_unc:\n        # 0.52 .. 0.95, ordered by the strongest single mention, nudged by repetition.\n        score = min(0.95, 0.50 + 0.42 * best + 0.03 * min(n_pos, 3))\n        conf = min(1.0, 0.55 + 0.15 * n_pos)\n    elif n_neg:\n        score = max(0.04, 0.20 - 0.04 * n_neg)\n        conf = min(0.9, 0.45 + 0.12 * n_neg)\n    else:\n        score, conf = 0.28, 0.05          # silence sits above asserted-negative\n    return score, conf, n_pos, n_neg\n\n\ndef extract(report: str) -> dict:\n    \"\"\"Extract twelve (score, confidence) pairs from one report.\"\"\"\n    cls = clauses(report)\n    out = {}\n    path_paired = _rx(TEAR.pattern, DEGEN.pattern, INJURY.pattern)\n\n    for tgt in TARGETS:\n        if tgt in PAIRED:\n            s, c, npos, nneg = _score_clauses(cls, ANAT_MATCH[tgt], path_paired)\n        elif tgt in OA_TARGETS:\n            s, c, npos, nneg = _score_clauses(cls, COMPARTMENT_MATCH[tgt], OA_EVIDENCE)\n        elif tgt == \"Contusion\":\n            # Reactive subchondral oedema under a cartilage defect is osteoarthritis,\n            # not a bruise. Explicit trauma wording pushes the other way.\n            s, c, npos, nneg = _score_clauses(cls, DIRECT_MATCH[tgt], None, DECOY.get(tgt),\n                                              context_penalty=DEGENERATIVE_MARROW,\n                                              context_bonus=TRAUMA)\n        else:\n            s, c, npos, nneg = _score_clauses(cls, DIRECT_MATCH[tgt], None, DECOY.get(tgt))\n        out[tgt] = s\n        out[tgt + \"__conf\"] = c\n        out[tgt + \"__npos\"] = npos\n        out[tgt + \"__nneg\"] = nneg\n\n    # --- cross-target corrections ------------------------------------------ #\n    # A whole-joint osteoarthritis statement is evidence for every compartment that was\n    # not separately assessed. Without this, \"incipient OA of all three compartments\"\n    # scores zero on all three OA targets.\n    g_hits = [c for c in cls if GLOBAL_OA.search(c) and _polarity(c) == \"positive\"]\n    if g_hits:\n        gscore = 0.50 + 0.42 * max(_severity(c) for c in g_hits)\n        for tgt in OA_TARGETS:\n            if out[tgt + \"__npos\"] == 0 and out[tgt + \"__nneg\"] == 0:\n                out[tgt] = max(out[tgt], gscore * 0.92)\n                out[tgt + \"__conf\"] = max(out[tgt + \"__conf\"], 0.4)\n\n    # Synovitis is frequently visible on the images and absent from the text, so silence\n    # is weak evidence of absence here in a way it is not for other findings. Effusion is\n    # its most reliable textual proxy - the two share a mechanism - so a silent synovitis\n    # inherits a fraction of the effusion evidence instead of falling to the floor.\n    if out[\"Synovitis__npos\"] == 0 and out[\"Synovitis__nneg\"] == 0:\n        out[\"Synovitis\"] = max(out[\"Synovitis\"], 0.28 + 0.45 * (out[\"Effusion\"] - 0.28))\n\n    return out","metadata":{"execution":{"iopub.status.busy":"2026-08-07T08:19:40.479903Z","iopub.execute_input":"2026-08-07T08:19:40.480389Z","iopub.status.idle":"2026-08-07T08:19:40.543843Z","shell.execute_reply.started":"2026-08-07T08:19:40.480357Z","shell.execute_reply":"2026-08-07T08:19:40.542368Z"},"papermill":{"duration":0.068453,"end_time":"2026-08-07T04:07:50.402715+00:00","exception":false,"start_time":"2026-08-07T04:07:50.334262+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#\n# This is the part the source notebooks do not have, and it exists to patch the one\n# failure the rule extractor cannot see from inside itself.\n#\n# `rsna-knee-baseline-v1` names the problem precisely: a rule that never fires emits a\n# negative rather than an error, so a lexicon thin in one language looks like a\n# population with fewer findings, and language tracks site. `rsna-knee-eda-to-2-5d`\n# fits a TF-IDF teacher, but on the 58 annotated studies only - far too few rows for a\n# 190k-feature model, which is why its reported OOF AUC sits near chance on several\n# targets.\n#\n# Fitting the text model to the *rule scores* instead uses all 4,407 reports. The model\n# then learns which character n-grams co-occur with rule-positive reports, in every\n# language present, and can score a Greek report the lexicon was silent on. Out-of-fold\n# and grouped on report text so it cannot memorise its own training signal.\n\ndef distill(reports, rule_scores, groups, n_splits=5, seed=2026, alpha=1.0,\n            verbose=True):\n    \"\"\"Fit a char+word TF-IDF ridge to the rule scores, out-of-fold.\n\n    reports      : list[str], length N\n    rule_scores  : (N, 12) float, the graded rule output\n    groups       : (N,) int, fold assignment (hash of report text - duplicates together)\n    returns      : (N, 12) float, OOF text predictions on the same 0-1 scale\n\n    Ridge on the graded score rather than logistic on a binarisation, because §1 of the\n    source notebook is right that only order is read and grading carries more signal\n    than a threshold does.\n    \"\"\"\n    from sklearn.feature_extraction.text import TfidfVectorizer\n    from sklearn.linear_model import Ridge\n    from scipy import sparse\n\n    texts = [normalize(r) for r in reports]\n    rule_scores = np.asarray(rule_scores, dtype=np.float32)\n    groups = np.asarray(groups)\n\n    # char_wb 3-5 is what makes this multilingual: it needs no tokeniser, no stemmer and\n    # no stopword list, so Greek and Turkish are handled on the same footing as English.\n    char = TfidfVectorizer(analyzer=\"char_wb\", ngram_range=(3, 5), min_df=3,\n                           max_features=200_000, sublinear_tf=True)\n    word = TfidfVectorizer(ngram_range=(1, 2), min_df=2, max_features=80_000,\n                           sublinear_tf=True)\n    X = sparse.hstack([char.fit_transform(texts), word.fit_transform(texts)],\n                      format=\"csr\")\n\n    oof = np.zeros_like(rule_scores)\n    folds = sorted(set(groups.tolist()))\n    for f in folds:\n        va = np.flatnonzero(groups == f)\n        tr = np.flatnonzero(groups != f)\n        if len(tr) < 20 or len(va) == 0:\n            oof[va] = rule_scores[va]\n            continue\n        m = Ridge(alpha=alpha, solver=\"sparse_cg\")\n        m.fit(X[tr], rule_scores[tr])\n        oof[va] = m.predict(X[va])\n\n    oof = np.clip(oof, 0.02, 0.98)\n\n    # How well the text model reproduces the rules out of fold, per target. This is the\n    # gate, not a diagnostic: the text model earns its weight by disagreeing with the\n    # rules where the rules are silent, but a model that cannot even recover them\n    # out-of-fold has learned nothing and must not be blended in. Targets whose lexicon\n    # is thin, or whose positives are too rare for the ridge to find, land here.\n    corr = np.zeros(rule_scores.shape[1], np.float32)\n    for j in range(rule_scores.shape[1]):\n        if np.std(oof[:, j]) > 1e-8 and np.std(rule_scores[:, j]) > 1e-8:\n            corr[j] = float(np.corrcoef(oof[:, j], rule_scores[:, j])[0, 1])\n    if verbose:\n        print(\"distilled text model, OOF corr with rules per target:\")\n        for t, a in zip(TARGETS, corr):\n            print(f\"  {t:<18} {a:+.3f}{'' if a > 0.15 else '   (gated out)'}\")\n    return oof, corr\n\n\ndef blend(rule_scores, text_scores, w_text=0.35):\n    \"\"\"Combine rules and the distilled text model in rank space, on the rules' scale.\n\n    Two steps, and the second is the one that is easy to get wrong.\n\n    *Combine as ranks.* The two sources are on different scales - the rules emit a\n    handful of discrete levels, the ridge a continuous spread - so averaging the values\n    lets the rules' coarse quantisation dominate. Ranks are the common currency.\n\n    *Map back onto the rules' scale, by interpolation.* The combined rank is uniform on\n    [0, 1] by construction, and using it directly as a BCE target would tell the model\n    that exactly half of all knees have a fracture. So it is read back through the rule\n    scores - but through their *distinct levels*, interpolating between them, not\n    through the sorted array.\n\n    Snapping to the sorted array instead is the obvious version and it silently undoes\n    the whole exercise. The rule score takes about four distinct values, so most of its\n    mass sits in one enormous atom at \"silent\"; a rank that lands anywhere inside that\n    atom snaps back to the same number, and every ordering the text model contributed\n    within the silent reports - which is precisely the ordering it was added to supply -\n    is quantised away. Interpolating between the levels keeps the prevalence and the\n    meaning of 0.5 while letting the text model order the reports inside each level.\n    \"\"\"\n    import pandas as pd\n\n    rule_scores = np.asarray(rule_scores, dtype=np.float32)\n    r = pd.DataFrame(rule_scores).rank(pct=True, method=\"average\").values\n    t = pd.DataFrame(text_scores).rank(pct=True, method=\"average\").values\n\n    # `w_text` may be a scalar or one weight per target - see `distill`, which returns\n    # the per-target OOF correlation used to gate it.\n    w = np.broadcast_to(np.asarray(w_text, np.float64).ravel(),\n                        (rule_scores.shape[1],)) if np.ndim(w_text) else \\\n        np.full(rule_scores.shape[1], float(w_text))\n    combined = (1.0 - w) * r + w * t\n\n    out = np.empty_like(rule_scores)\n    n = len(rule_scores)\n    for j in range(rule_scores.shape[1]):\n        levels, counts = np.unique(rule_scores[:, j], return_counts=True)\n        if len(levels) == 1:\n            out[:, j] = levels[0]\n            continue\n        # Anchor each distinct level at the midpoint of the rank interval it occupies,\n        # then interpolate. Anchoring at the midpoint rather than an edge keeps the map\n        # centred on the level, so a report the rules scored at that level and the text\n        # model had no opinion about comes back out where it went in.\n        mid = (np.cumsum(counts) - counts / 2.0) / n\n        out[:, j] = np.interp(combined[:, j], mid, levels)\n    return out","metadata":{"execution":{"iopub.status.busy":"2026-08-07T08:19:40.597366Z","iopub.execute_input":"2026-08-07T08:19:40.598497Z","iopub.status.idle":"2026-08-07T08:19:40.616298Z","shell.execute_reply.started":"2026-08-07T08:19:40.598459Z","shell.execute_reply":"2026-08-07T08:19:40.615156Z"},"papermill":{"duration":0.019653,"end_time":"2026-08-07T04:07:50.435473+00:00","exception":false,"start_time":"2026-08-07T04:07:50.41582+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"RSNA knee: study-level 12-label pipeline.\n\nSynthesised from two community notebooks, with the parts that were wrong or missing in\nboth replaced. What is inherited, and from where:\n\n  from `rsna-knee-baseline-v1`   header-recovered slot scheme, constant-physical-scale\n                                 sampling, laterality normalisation, uint8 cache,\n                                 per-diagnosis slot attention, report-hash grouping\n  from `rsna-knee-eda-to-2-5d`   protocol-only features as a legal test-time signal\n\nWhat is new here is listed in `README.md`; the five that change results are grouped\nK-fold instead of a single holdout, gold studies held out of their own fold so the\nannotated reference is honest, anatomy-preserving augmentation, a backbone that refuses\nto train from random initialisation silently, and rank fusion of the folds and the\nprotocol model.\n\"\"\"\n\nfrom __future__ import annotations\n\nimport os\n\nfor _v in (\"OMP_NUM_THREADS\", \"OPENBLAS_NUM_THREADS\", \"MKL_NUM_THREADS\"):\n    os.environ.setdefault(_v, \"4\")\n\nimport gc\nimport hashlib\nimport re\nimport time\nimport warnings\nfrom concurrent.futures import ThreadPoolExecutor\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\n\nT0 = time.time()\n\n\ndef log(msg):\n    print(f\"[{time.time() - T0:7.1f}s] {msg}\", flush=True)","metadata":{"execution":{"iopub.status.busy":"2026-08-07T08:19:40.618448Z","iopub.execute_input":"2026-08-07T08:19:40.618874Z","iopub.status.idle":"2026-08-07T08:19:45.381556Z","shell.execute_reply.started":"2026-08-07T08:19:40.618847Z","shell.execute_reply":"2026-08-07T08:19:45.380492Z"},"papermill":{"duration":4.659378,"end_time":"2026-08-07T04:07:55.108565+00:00","exception":false,"start_time":"2026-08-07T04:07:50.449187+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n    seed = 2026\n\n    img = 224                 # encoder input; a multiple of the DINOv2 patch size 14\n    crop_mm = 160.0           # physical extent of the centre crop; a knee FOV is 140-180\n    group = 3                 # slices per encoder input, stacked as the three channels\n    slice_stride = 2          # stride between 2.5D triplet slices across Z-axis\n    slice_jitter = True       # random slice anchor jitter during training\n    n_group_max = 3           # groups per slot, before the memory budget is applied\n    cache_budget_gb = 12.0\n    ram_fraction = 0.45       # ceiling on the cache as a share of free RAM\n\n    hdr_threads = 16\n    pix_threads = 12\n\n    n_folds = 5\n    max_folds_to_run = int(os.environ.get(\"RSNA_FOLDS\", \"5\"))\n    epochs = 12\n    batch_studies = 8\n    lr_backbone = 8e-6        # the encoder is being adapted, not retrained\n    lr_head = 1e-3\n    weight_decay = 0.02\n    unfreeze_last = 6         # trainable transformer blocks, from the output end\n    eval_batch = 12\n    time_budget = float(os.environ.get(\"RSNA_TIME_BUDGET\", 8.0 * 3600))\n\n    w_text = 0.35             # weight of the distilled text model in the label blend\n    w_protocol = 0.10         # weight of the protocol-only model in the final fusion\n    gold_weight = 3.0\n\n\n# Six slots: three planes crossed with the acquisition axes. The fat-suppressed\n# fluid-sensitive series exist for nearly every study; the T1 and the non-suppressed\n# fluid-sensitive series are scarcer, which is what the presence mask is for.\nSLOTS_RECOVERED = [\n    (\"SAG_FLUID_FS\", \"Sagittal\", True, True),\n    (\"COR_FLUID_FS\", \"Coronal\", True, True),\n    (\"AX_FLUID_FS\", \"Axial\", True, True),\n    (\"SAG_FLUID_NOFS\", \"Sagittal\", True, False),\n    (\"COR_T1\", \"Coronal\", False, False),\n    (\"SAG_T1\", \"Sagittal\", False, False),\n]\n\n# The alternative: plane x the provided flag, ignoring the recovered weighting.\n#\n# Worth stating why the recovered scheme is the default. The EDA notebook found that\n# `Fluid_Sensitive` and `Fat_Suppression` are byte-identical across all 24,371 series\n# rows. They name two physically independent properties - contrast weighting, set by\n# TR/TE, and a fat-suppression preparation applied on top of any weighting - so two\n# identical columns carry one bit between them, not two. Recovering both axes from the\n# DICOM header is therefore not a refinement; it is the difference between six slots\n# and three.\nSLOTS_PUBLIC = [\n    (\"SAG_FLUID\", \"Sagittal\", None, True),\n    (\"COR_FLUID\", \"Coronal\", None, True),\n    (\"AX_FLUID\", \"Axial\", None, True),\n    (\"SAG_STRUCT\", \"Sagittal\", None, False),\n    (\"COR_STRUCT\", \"Coronal\", None, False),\n    (\"AX_STRUCT\", \"Axial\", None, False),\n]\n\nSLOT_SCHEME = os.environ.get(\"SLOT_SCHEME\", \"recovered\")\nSLOTS = SLOTS_PUBLIC if SLOT_SCHEME == \"public\" else SLOTS_RECOVERED\nN_SLOT = len(SLOTS)\n\nFATSAT_OPTS = {\"FS\", \"FATSAT\", \"FAT_SAT\", \"FSAT\"}\n_SEP = re.compile(r\"[_\\-.]\")\n_FATSAT_RX = re.compile(r\"\\bfs\\b|fatsat|fat sat|\\bstir\\b|\\bspair\\b|\\bspir\\b|\\bwe\\b|\"\n                        r\"water excit|\\btirm\\b|\\bsting\\b|\\bfatsup\\b\")\n_T1_RX = re.compile(r\"\\bt1\\b|\\bt1w\\b\")\n_T2_RX = re.compile(r\"\\bt2\\b|\\bt2w\\b\")\n_PD_RX = re.compile(r\"\\bpd\\b|\\bpdw\\b|proton|\\bdp\\b|dens\")\n\n\ndef seed_all(seed=CFG.seed):\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)","metadata":{"execution":{"iopub.status.busy":"2026-08-07T08:19:45.382989Z","iopub.execute_input":"2026-08-07T08:19:45.383705Z","iopub.status.idle":"2026-08-07T08:19:45.398605Z","shell.execute_reply.started":"2026-08-07T08:19:45.383654Z","shell.execute_reply":"2026-08-07T08:19:45.397467Z"},"papermill":{"duration":0.028242,"end_time":"2026-08-07T04:07:55.145582+00:00","exception":false,"start_time":"2026-08-07T04:07:55.11734+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def find_root(explicit=None):\n    if explicit is not None:\n        p = Path(explicit)\n        if (p / \"test.csv\").is_file():\n            return p\n        raise FileNotFoundError(f\"no test.csv under {p}\")\n    for c in [Path(\"/kaggle/input/competitions/rsna-knee-abnormality-detection\"),\n              Path(\"/kaggle/input/rsna-knee-abnormality-detection\"),\n              Path(\"data\"), Path(\".\")]:\n        if (c / \"test.csv\").is_file() and (c / \"test_series\").is_dir():\n            return c\n    base = Path(\"/kaggle/input\")\n    if base.is_dir():\n        # last resort: two-level scan, because the mount is sometimes nested one deeper\n        for depth1 in sorted(p for p in base.iterdir() if p.is_dir()):\n            for cand in [depth1] + sorted(p for p in depth1.iterdir() if p.is_dir()):\n                if (cand / \"test.csv\").is_file():\n                    return cand\n    raise FileNotFoundError(\"competition mount not found\")\n\n\ndef find_dinov2(variant=\"small\"):\n    \"\"\"Locate a mounted DINOv2 checkpoint directory by variant name.\"\"\"\n    base = Path(\"/kaggle/input\")\n    if not base.is_dir():\n        return None\n    hits = []\n    for root, dirs, files in os.walk(base):\n        dirs[:] = [d for d in dirs if d not in (\"train_series\", \"test_series\")]\n        if \"config.json\" in files and \"dinov2\" in root.lower():\n            hits.append(Path(root))\n    for h in hits:\n        if variant in str(h).lower():\n            return h\n    return hits[0] if hits else None","metadata":{"execution":{"iopub.status.busy":"2026-08-07T08:19:45.399927Z","iopub.execute_input":"2026-08-07T08:19:45.400313Z","iopub.status.idle":"2026-08-07T08:19:45.430953Z","shell.execute_reply.started":"2026-08-07T08:19:45.400274Z","shell.execute_reply":"2026-08-07T08:19:45.429425Z"},"papermill":{"duration":0.013524,"end_time":"2026-08-07T04:07:55.165088+00:00","exception":false,"start_time":"2026-08-07T04:07:55.151564+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"HDR_TAGS = [\"SeriesDescription\", \"SequenceName\", \"ScanOptions\", \"ScanningSequence\",\n            \"RepetitionTime\", \"EchoTime\", \"Laterality\", \"PixelSpacing\", \"Rows\",\n            \"Columns\", \"RescaleSlope\", \"RescaleIntercept\"]\n\n\ndef probe(item):\n    split, study, series, path = item\n    row = {\"split\": split, \"StudyInstanceUID\": study, \"SeriesInstanceUID\": series,\n           \"dir\": path}\n    try:\n        files = sorted(e.name for e in os.scandir(path) if e.name.endswith(\".dcm\"))\n        row[\"files\"] = files\n        row[\"n_slices\"] = len(files)\n        if not files:\n            return row\n        ds = pydicom.dcmread(os.path.join(path, files[len(files) // 2]),\n                             stop_before_pixels=True, force=True)\n        for t in HDR_TAGS:\n            v = getattr(ds, t, None)\n            if v is None:\n                row[t] = None\n            elif isinstance(v, (list, tuple)) or type(v).__name__ == \"MultiValue\":\n                row[t] = \"|\".join(str(x) for x in v)\n            else:\n                row[t] = str(v)\n    except Exception as exc:\n        row[\"err\"] = str(exc)[:120]\n    return row\n\n\ndef walk(root, split):\n    base = Path(root) / split\n    items = []\n    if not base.is_dir():\n        return pd.DataFrame()\n    for study in os.scandir(base):\n        if study.is_dir():\n            for series in os.scandir(study.path):\n                if series.is_dir():\n                    items.append((split, study.name, series.name, series.path))\n    with ThreadPoolExecutor(max_workers=CFG.hdr_threads) as pool:\n        rows = list(pool.map(probe, items))\n    return pd.DataFrame(rows)\n\n\ndef annotate(df):\n    \"\"\"Recover fat suppression and pulse-sequence weighting from the header.\n\n    Two string-matching cautions, both of which silently invert the answer if missed:\n    underscore is a word character, so a token test for `we` (water excitation) never\n    fires inside `t2_de3d_we_tra` unless separators are normalised first; and GE writes\n    `SAT_GEMS` for *spatial* saturation, so ScanOptions must be matched as exact tokens\n    or non-fat-suppressed series get marked as suppressed.\n    \"\"\"\n    if df.empty:\n        return df\n    for t in HDR_TAGS:\n        if t not in df.columns:\n            df[t] = None\n\n    desc = (df[\"SeriesDescription\"].fillna(\"\") + \" \" + df[\"SequenceName\"].fillna(\"\"))\n    desc = desc.str.lower().str.replace(_SEP, \" \", regex=True)\n\n    opts = df[\"ScanOptions\"].fillna(\"\").str.upper().str.split(\"|\")\n    opts_fs = opts.apply(lambda ts: any(t.strip() in FATSAT_OPTS for t in ts))\n    df[\"fatsat\"] = desc.str.contains(_FATSAT_RX) | opts_fs\n\n    tr = pd.to_numeric(df[\"RepetitionTime\"], errors=\"coerce\")\n    te = pd.to_numeric(df[\"EchoTime\"], errors=\"coerce\")\n    gre = df[\"ScanningSequence\"].fillna(\"\").str.upper().str.contains(\"GR\")\n    t1 = desc.str.contains(_T1_RX)\n    t2 = desc.str.contains(_T2_RX)\n    pdw = desc.str.contains(_PD_RX)\n\n    df[\"weight\"] = np.where(t1 & ~t2 & ~pdw, \"T1\",\n                     np.where(t2 & ~pdw, \"T2\",\n                       np.where(pdw, \"PD\",\n                         np.where(gre, \"GRE\",\n                           np.where(tr < 800, \"T1\",\n                             np.where(te > 60, \"T2\",\n                               np.where(tr >= 800, \"PD\", \"UNK\")))))))\n    df[\"fluid\"] = np.isin(df[\"weight\"], [\"PD\", \"T2\"])\n    df[\"px\"] = pd.to_numeric(\n        df[\"PixelSpacing\"].fillna(\"\").str.split(\"|\").str[0].replace(\"\", np.nan),\n        errors=\"coerce\")\n    return df\n\n\ndef pick_slots(series_df, plane_map):\n    \"\"\"One series per slot per study.\n\n    Ties are broken toward the stack with the most slices: a thicker stack samples the\n    joint more densely, and the three-slice sampler benefits from the margin.\n    \"\"\"\n    if series_df.empty:\n        return {}\n    series_df = series_df.copy()\n    series_df[\"plane\"] = series_df[\"SeriesInstanceUID\"].map(plane_map)\n    out = {}\n    for study, g in series_df.groupby(\"StudyInstanceUID\"):\n        chosen = {}\n        for name, plane, fluid, fs in SLOTS:\n            sel = (g[\"plane\"] == plane) & (g[\"fatsat\"] == fs)\n            # fluid=None means \"do not condition on weighting\" - the public scheme,\n            # where the single provided flag stands in for both axes at once.\n            if fluid is not None:\n                sel = sel & (g[\"fluid\"] == fluid)\n            cand = g[sel]\n            if len(cand) == 0 and fluid is False:\n                # T1 slots are the scarcest; fall back to any non-fat-sat series in the\n                # plane before giving up on the slot entirely.\n                cand = g[(g[\"plane\"] == plane) & (~g[\"fatsat\"])]\n            if len(cand):\n                chosen[name] = cand.sort_values(\"n_slices\", ascending=False).iloc[0]\n        out[study] = chosen\n    return out","metadata":{"execution":{"iopub.status.busy":"2026-08-07T08:19:45.433456Z","iopub.execute_input":"2026-08-07T08:19:45.433942Z","iopub.status.idle":"2026-08-07T08:19:45.46333Z","shell.execute_reply.started":"2026-08-07T08:19:45.433912Z","shell.execute_reply":"2026-08-07T08:19:45.462091Z"},"papermill":{"duration":0.022223,"end_time":"2026-08-07T04:07:55.191764+00:00","exception":false,"start_time":"2026-08-07T04:07:55.169541+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ORDER_TAGS = [\"ImagePositionPatient\", \"ImageOrientationPatient\", \"SliceLocation\",\n              \"InstanceNumber\"]\n\n# Ordering happens on reader threads, so `d[k] += 1` - three bytecodes, not one - needs\n# the lock. An undercounted fallback reads as a clean run.\nimport threading\n\nORDER_STATS = {\"geometry\": 0, \"slice_location\": 0, \"instance_number\": 0, \"filename\": 0}\n_ORDER_LOCK = threading.Lock()\n\n\ndef _bump(key):\n    with _ORDER_LOCK:\n        ORDER_STATS[key] += 1\n\n\ndef order_slices(directory, files):\n    \"\"\"Return `files` sorted by position along the stack, nearest-first.\n\n    The filenames in this corpus are SOP Instance UIDs, which are random, so the sorted\n    filename order is a random permutation of the anatomy. Measured on the competition\n    data, filename order agrees with physical order on about 5% of slices.\n\n    That is not a cosmetic problem. Two things downstream assume the list is ordered:\n    the sampling window (\"the central portion of the stack\") and the grouping of\n    adjacent slices into the encoder's three channels. Under a random permutation the\n    window selects a uniform random sample of the whole series rather than its centre,\n    and a group of three adjacent entries is three unrelated positions in the knee - so\n    the 2.5D input carried no depth information at all.\n\n    Fallback chain, because not every series records the same tags:\n      1. project ImagePositionPatient onto the normal of ImageOrientationPatient. This\n         is the only one that is correct for an obliquely angled stack.\n      2. SliceLocation, which the scanner has already projected but which is signed\n         inconsistently between vendors.\n      3. InstanceNumber, acquisition order - right for most stacks, wrong for\n         interleaved acquisitions.\n      4. filename, i.e. give up, and say so in the counters.\n    \"\"\"\n    if len(files) < 2:\n        return list(files)\n\n    positions, locations, instances = [], [], []\n    normal = None\n    for name in files:\n        try:\n            ds = pydicom.dcmread(os.path.join(directory, name), stop_before_pixels=True,\n                                 force=True, specific_tags=ORDER_TAGS)\n        except Exception:\n            positions.append(None)\n            locations.append(None)\n            instances.append(None)\n            continue\n        pos = getattr(ds, \"ImagePositionPatient\", None)\n        orient = getattr(ds, \"ImageOrientationPatient\", None)\n        if normal is None and orient is not None and len(orient) == 6:\n            try:\n                row_dir = np.array([float(v) for v in orient[:3]])\n                col_dir = np.array([float(v) for v in orient[3:]])\n                normal = np.cross(row_dir, col_dir)\n            except Exception:\n                normal = None\n        try:\n            positions.append(np.array([float(v) for v in pos]) if pos is not None else None)\n        except Exception:\n            positions.append(None)\n        loc = getattr(ds, \"SliceLocation\", None)\n        locations.append(float(loc) if loc is not None else None)\n        num = getattr(ds, \"InstanceNumber\", None)\n        instances.append(float(num) if num is not None else None)\n\n    def sorted_by(values, stat):\n        _bump(stat)\n        return [f for _, f in sorted(zip(values, files), key=lambda pair: pair[0])]\n\n    if normal is not None and all(p is not None for p in positions):\n        return sorted_by([float(p @ normal) for p in positions], \"geometry\")\n    if all(loc is not None for loc in locations):\n        return sorted_by(locations, \"slice_location\")\n    if all(num is not None for num in instances):\n        return sorted_by(instances, \"instance_number\")\n    _bump(\"filename\")\n    return list(files)\n\n\n# Fraction of the ordered stack each plane is sampled across.\n#\n# These are wider than the 20-80% window this replaced, and the widening is part of the\n# ordering fix rather than a separate change. Before ordering, \"the central 60%\" of a\n# random permutation was a uniform sample of the *whole* series; applying that same\n# window to a correctly ordered stack would genuinely discard the outer 40% and quietly\n# bundle a coverage reduction into this change. These keep the physical span roughly as\n# it was, so the ordering is the only variable that moved.\n#\n# The per-plane split follows Will's (`wguesdon`) reasoning: the menisci sit at the\n# medial and lateral extremes of a sagittal stack and Baker's cysts sit posteriorly on\n# an axial one, so those planes need their edges; a coronal stack's useful\n# anterior-posterior range is narrower.\nPLANE_WINDOW = {\"Sagittal\": (0.10, 0.90), \"Axial\": (0.10, 0.90),\n                \"Coronal\": (0.15, 0.85)}\n\n\ndef read_slot(rec, n_slice, out_size, plane=None, group=None, stride=None, jitter=False):\n    \"\"\"`n_slice` slices from one series, physically ordered, at `out_size` pixels.\n\n    Returns uint8 [n_slice, out, out], normalised per-series to its 1st-99th percentile.\n    Percentiles rather than min/max because MR intensity has no absolute scale and one\n    bright vessel would otherwise compress the whole dynamic range.\n\n    The slices come out as `n_slice // group` anchors spread across the sampling window,\n    each anchor contributing `group` *physically adjacent* slices. That layout is what\n    `take_group` slices back out into the encoder's three channels, so a channel triplet\n    is a genuine depth neighbourhood - roughly 10 mm of knee at this corpus's slice gaps\n    - rather than three arbitrary positions. Spreading all nine slices evenly instead\n    would cover more of the joint but hand the encoder three views 20 mm apart, which is\n    a slab, not a 2.5D triplet.\n\n    A slice of N pixels at spacing s mm covers Ns millimetres. Both vary widely here, so\n    a fixed-pixel resize hands the encoder images whose physical scale differs by about\n    3x. Cropping to a constant physical extent first fixes the effective scale at\n    crop_mm / img mm per pixel regardless of acquisition.\n    \"\"\"\n    files, d, px = rec[\"files\"], rec[\"dir\"], rec[\"px\"]\n    n = len(files)\n    if n == 0:\n        return None\n    group = CFG.group if group is None else group\n    stride = getattr(CFG, 'slice_stride', 2) if stride is None else stride\n\n    files = order_slices(d, files)\n\n    lo_f, hi_f = PLANE_WINDOW.get(plane, (0.10, 0.90))\n    lo, hi = int(lo_f * (n - 1)), int(hi_f * (n - 1))\n    if hi <= lo:\n        lo, hi = 0, n - 1\n\n    n_anchor = max(1, n_slice // group)\n    anchors = (np.linspace(lo, hi, n_anchor).astype(int) if n_anchor > 1\n               else np.array([(lo + hi) // 2]))\n\n    idx = []\n    for centre in anchors:\n        if jitter and getattr(CFG, 'slice_jitter', True):\n            centre = centre + int(np.random.randint(-1, 2))\n            centre = int(np.clip(centre, 0, n - 1))\n        offset = (group // 2) * stride\n        start = int(np.clip(centre - offset, 0, max(0, n - 1 - (group - 1) * stride)))\n        indices = [int(np.clip(start + k * stride, 0, n - 1)) for k in range(group)]\n        idx.extend(indices)\n    while len(idx) < n_slice:\n        idx.append(idx[-1])\n\n    planes = []\n    for i in idx[:n_slice]:\n        try:\n            ds = pydicom.dcmread(os.path.join(d, files[int(i)]), force=True)\n            a = ds.pixel_array.astype(np.float32)\n            sl = float(getattr(ds, \"RescaleSlope\", 1) or 1)\n            ic = float(getattr(ds, \"RescaleIntercept\", 0) or 0)\n            a = a * sl + ic\n        except Exception:\n            a = None\n        planes.append(a)\n\n    shp = next((p.shape for p in planes if p is not None), None)\n    if shp is None:\n        return None\n    planes = [p if (p is not None and p.shape == shp) else np.zeros(shp, np.float32)\n              for p in planes]\n    vol = np.stack(planes)\n\n    if px and np.isfinite(px) and px > 0:\n        want = int(round(CFG.crop_mm / px))\n        h, w = shp\n        if 16 < want < min(h, w):\n            cy, cx = h // 2, w // 2\n            half = want // 2\n            vol = vol[:, max(0, cy - half):cy + half, max(0, cx - half):cx + half]\n\n    lo_v, hi_v = np.percentile(vol, [1, 99])\n    vol = np.clip((vol - lo_v) / max(hi_v - lo_v, 1e-6), 0, 1)\n\n    t = torch.from_numpy(np.ascontiguousarray(vol)).unsqueeze(0)\n    t = F.interpolate(t, size=(out_size, out_size), mode=\"bilinear\", align_corners=False)\n    # uint8, not float32: intensity is already normalised into [0, 1] here, so eight bits\n    # cost nothing a bilinear resize has not already cost, and the cache is a quarter the\n    # size for it.\n    return (t.squeeze(0) * 255).round().clamp(0, 255).to(torch.uint8)\n\n\ndef normalise_laterality(img, plane, lat):\n    \"\"\"Map every knee onto a left-knee convention.\n\n    Four of the twelve targets are medial/lateral pairs, and medial is defined against\n    the body midline, so which side of the *image* it falls on depends on which knee was\n    scanned. Coronal and axial views mirror under a horizontal flip. Sagittal stacks do\n    not - each slice is unchanged by mirroring, what differs is the direction the stack\n    traverses the joint - so the slice order is reversed instead.\n\n    Where `Laterality` is absent the volume is left alone: a wrong flip is worse than no\n    flip, and the presence mask lets the head learn how much to trust each slot.\n    \"\"\"\n    if lat != \"R\":\n        return img\n    if plane in (\"Coronal\", \"Axial\"):\n        return torch.flip(img, dims=[-1])\n    return torch.flip(img, dims=[0])\n\n\ndef available_ram_gb():\n    \"\"\"Physical RAM available to this process, in GB, or None if it cannot be read.\"\"\"\n    try:\n        import psutil\n        return psutil.virtual_memory().available / 1024 ** 3\n    except Exception:\n        pass\n    try:                                   # Linux without psutil, which includes Kaggle\n        with open(\"/proc/meminfo\") as fh:\n            for line in fh:\n                if line.startswith(\"MemAvailable:\"):\n                    return int(line.split()[1]) / 1024 ** 2\n    except Exception:\n        pass\n    return None\n\n\ndef plan_cache(n_study):\n    \"\"\"Choose how many slices per slot memory allows.\n\n    The cache is n_study x n_slot x slices x img^2 bytes. It grows with the *square* of\n    resolution and only linearly with slices, so coverage is the cheap axis and\n    resolution the expensive one; when the budget binds it is the slice count that gives\n    way. Deciding once, from the training corpus size, keeps train and test on the same\n    group layout.\n\n    The configured budget is a ceiling, not a promise. At full corpus size the cache is\n    around 12 GB, which fits a 30 GB Kaggle instance and does not fit a 13 GB one - and\n    the failure mode is a SIGKILL partway through the decode, which no `try` will catch\n    and which costs the whole run. So the budget is also capped at a fraction of what\n    the machine actually reports free, leaving room for the model, the reader queue and\n    the test cache.\n    \"\"\"\n    budget = CFG.cache_budget_gb\n    free = available_ram_gb()\n    if free is not None:\n        safe = CFG.ram_fraction * free\n        if safe < budget:\n            log(f\"cache budget trimmed {budget:.1f} -> {safe:.1f} GB \"\n                f\"({free:.1f} GB free, keeping {1 - CFG.ram_fraction:.0%} headroom)\")\n            budget = safe\n    per_slice = n_study * N_SLOT * CFG.img * CFG.img\n    afford = int(budget * 1024 ** 3 // max(per_slice, 1))\n    groups = max(1, min(CFG.n_group_max, afford // CFG.group))\n    if groups < CFG.n_group_max:\n        log(f\"cache budget {budget:.1f} GB allows {groups} group(s) of \"\n            f\"{CFG.group}, not {CFG.n_group_max}\")\n    return groups\n\n\ndef build_cache(slot_map, plane_map, lat_map, tag, n_group):\n    \"\"\"Decode every (study, slot) once into an in-memory uint8 array.\n\n    Fine-tuning revisits the same pixels every epoch, and a study is on the order of a\n    hundred and fifty files. Reading them from the mount each epoch would make the epoch\n    count a function of I/O rather than of learning.\n\n    Reads are issued in bounded chunks: submitting every job at once lets the reader\n    threads run arbitrarily far ahead and the completed buffers accumulate without limit.\n    \"\"\"\n    cache_slices = CFG.group * n_group\n    studies = sorted(slot_map)\n    sidx = {s: i for i, s in enumerate(studies)}\n    cache = np.zeros((len(studies), N_SLOT, cache_slices, CFG.img, CFG.img), np.uint8)\n    mask = np.zeros((len(studies), N_SLOT), np.float32)\n    log(f\"{tag}: cache {cache.shape} = {cache.nbytes / 1024 ** 3:.1f} GB\")\n\n    jobs = [(st, k, plane, slot_map[st][name])\n            for st in studies\n            for k, (name, plane, _, _) in enumerate(SLOTS)\n            if name in slot_map[st]]\n    log(f\"{tag}: decoding {len(jobs)} slot-series\")\n\n    chunk = 512\n    done = 0\n    with ThreadPoolExecutor(max_workers=CFG.pix_threads) as pool:\n        for c0 in range(0, len(jobs), chunk):\n            block = jobs[c0:c0 + chunk]\n            imgs = pool.map(lambda j: read_slot(j[3], cache_slices, CFG.img, j[2]), block)\n            for (st, k, plane, _), img in zip(block, imgs):\n                done += 1\n                if img is None:\n                    continue\n                cache[sidx[st], k] = normalise_laterality(img, plane,\n                                                          lat_map.get(st)).numpy()\n                mask[sidx[st], k] = 1.0\n            if done % 4096 < chunk:\n                log(f\"  {tag} {done}/{len(jobs)}\")\n            if time.time() - T0 > CFG.time_budget:\n                log(f\"  {tag}: time budget reached during decode\")\n                break\n    total = sum(ORDER_STATS.values()) or 1\n    log(f\"{tag}: slice ordering \" + \", \".join(\n        f\"{k} {v / total:.1%}\" for k, v in ORDER_STATS.items() if v))\n    if ORDER_STATS[\"filename\"] / total > 0.05:\n        warnings.warn(\n            f\"{ORDER_STATS['filename'] / total:.0%} of series fell back to filename \"\n            \"order, which is SOP UID order and therefore random. Those slots carry no \"\n            \"depth structure.\", RuntimeWarning, stacklevel=2)\n    gc.collect()\n    return studies, cache, mask\n","metadata":{"execution":{"iopub.status.busy":"2026-08-07T08:19:45.464669Z","iopub.execute_input":"2026-08-07T08:19:45.465028Z","iopub.status.idle":"2026-08-07T08:19:45.512723Z","shell.execute_reply.started":"2026-08-07T08:19:45.464999Z","shell.execute_reply":"2026-08-07T08:19:45.511555Z"},"papermill":{"duration":0.03534,"end_time":"2026-08-07T04:07:55.240288+00:00","exception":false,"start_time":"2026-08-07T04:07:55.204948+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class SlotHead(nn.Module):\n    \"\"\"Per-diagnosis attention over the slot embeddings of one study.\n\n    Each finding is read on particular sequences - cruciates sagittally, collateral\n    ligaments and the meniscal body coronally, patellar cartilage axially - so pooling\n    the slots identically would dilute the one that carries the evidence with the rest.\n    The softmax is masked, so a missing axial series shifts a diagnosis's attention onto\n    the sequences that are present instead of feeding it a zero vector.\n\n    The aggregation is deliberately this simple. The label is attached to the *study*, so\n    nothing in the supervision says which part of a study carries the finding; extra\n    attention parameters have no signal to learn that from and spend capacity on noise.\n    Where the supervision is coarse, the aggregation should be too.\n    \"\"\"\n\n    def __init__(self, dim, n_slot, n_out, hidden=256, p=0.2):\n        super().__init__()\n        self.proj = nn.Sequential(nn.LayerNorm(dim), nn.Linear(dim, hidden), nn.GELU())\n        self.slot_emb = nn.Parameter(torch.randn(n_slot, hidden) * 0.02)\n        self.query = nn.Parameter(torch.randn(n_out, hidden) * 0.02)\n        self.drop = nn.Dropout(p)\n        self.out = nn.Linear(hidden, n_out)\n        self.hidden = hidden\n\n    def forward(self, x, mask):\n        h = self.proj(x) + self.slot_emb\n        att = torch.einsum(\"bsh,oh->bos\", h, self.query) / self.hidden ** 0.5\n        att = att.masked_fill(mask.unsqueeze(1) < 0.5, -1e4).softmax(-1)\n        ctx = self.drop(torch.einsum(\"bos,bsh->boh\", att, h))\n        return (ctx * self.out.weight.unsqueeze(0)).sum(-1) + self.out.bias\n\n\nclass Encoder(nn.Module):\n    \"\"\"Uniform [B,3,H,W] -> [B,dim] wrapper over either a HF or a timm backbone.\"\"\"\n\n    def __init__(self, module, kind, dim):\n        super().__init__()\n        self.module = module\n        self.kind = kind\n        self.dim = dim\n\n    def forward(self, x):\n        if self.kind == \"hf\":\n            out = self.module(pixel_values=x).last_hidden_state\n            # CLS and mean-pooled patches: the CLS token carries the global summary and\n            # the patch mean the spatially distributed evidence, and the findings here\n            # need both.\n            return torch.cat([out[:, 0], out[:, 1:].mean(1)], dim=1)\n        return self.module(x)\n\n    def blocks(self):\n        if self.kind == \"hf\":\n            return list(self.module.encoder.layer)\n        for attr in (\"blocks\", \"layers\"):\n            b = getattr(self.module, attr, None)\n            if b is not None:\n                return list(b)\n        return []\n\n\nclass Model(nn.Module):\n    \"\"\"Encoder plus head, trained end to end.\n\n    A study arrives as a bag of slot images. The bag is flattened for the encoder and\n    folded back before the head, so the encoder never sees the study structure and the\n    head never sees pixels.\n    \"\"\"\n\n    def __init__(self, encoder):\n        super().__init__()\n        self.encoder = encoder\n        self.head = SlotHead(encoder.dim, N_SLOT, len(TARGETS))\n        self.register_buffer(\"mean\", torch.tensor([0.485, 0.456, 0.406]).view(1, 3, 1, 1))\n        self.register_buffer(\"std\", torch.tensor([0.229, 0.224, 0.225]).view(1, 3, 1, 1))\n\n    def forward(self, imgs, mask):\n        b, s = imgs.shape[:2]\n        x = imgs.reshape(b * s, *imgs.shape[2:]).float().div_(255.0)\n        x = (x - self.mean) / self.std\n        feat = self.encoder(x).reshape(b, s, -1)\n        return self.head(feat, mask)\n\n\ndef build_encoder():\n    \"\"\"Resolve a backbone, and refuse to train from random initialisation silently.\n\n    Order: an explicit RSNA_BACKBONE timm name, then a mounted DINOv2 directory, then\n    timm with whatever cached weights exist. A frozen self-supervised encoder saturates\n    on this task - resolution, encoder size, slice coverage and slot aggregation all\n    stop moving the score at the same place, which says the binding constraint is the\n    representation, learned as it was on natural images where nothing resembles a torn\n    meniscus on a proton-density sequence. So the last blocks are adapted.\n\n    The random-initialisation path stays available but shouts. One of the two source\n    notebooks builds `resnet18` with `pretrained=False` and blends 88% of its output\n    into the submission, which is a randomly initialised network given two epochs on\n    4,407 studies; that is the failure this warning exists to make impossible to miss.\n    \"\"\"\n    name = os.environ.get(\"RSNA_BACKBONE\")\n    if name:\n        import timm\n        m = timm.create_model(name, pretrained=True, num_classes=0, global_pool=\"avg\")\n        log(f\"backbone: timm {name}, dim {m.num_features}\")\n        enc = Encoder(m, \"timm\", m.num_features)\n    else:\n        p = find_dinov2(\"small\")\n        if p is not None:\n            from transformers import AutoModel\n            bb = AutoModel.from_pretrained(str(p))\n            enc = Encoder(bb, \"hf\", bb.config.hidden_size * 2)\n            log(f\"backbone: DINOv2 from {p}, dim {enc.dim}\")\n        else:\n            import timm\n            fallback = \"resnet18\"\n            try:\n                m = timm.create_model(fallback, pretrained=True, num_classes=0,\n                                      global_pool=\"avg\")\n                log(f\"backbone: timm {fallback} (pretrained), dim {m.num_features}\")\n            except Exception as exc:\n                m = timm.create_model(fallback, pretrained=False, num_classes=0,\n                                      global_pool=\"avg\")\n                warnings.warn(\n                    \"NO PRETRAINED WEIGHTS AVAILABLE - training from random \"\n                    f\"initialisation ({exc}). On 4,407 weakly labelled studies this \"\n                    \"will not learn anything useful. Attach a DINOv2 or timm weights \"\n                    \"dataset before trusting any score from this run.\",\n                    RuntimeWarning, stacklevel=2)\n                log(\"!! backbone: RANDOM INIT - results are not meaningful\")\n            enc = Encoder(m, \"timm\", m.num_features)\n\n    for prm in enc.parameters():\n        prm.requires_grad = False\n    blocks = enc.blocks()\n    if blocks:\n        # The early blocks of a self-supervised transformer are generic edge and texture\n        # filters. There is not enough supervision here to improve them and quite enough\n        # to damage them, so only the last few move.\n        for blk in blocks[max(0, len(blocks) - CFG.unfreeze_last):]:\n            for prm in blk.parameters():\n                prm.requires_grad = True\n        if enc.kind == \"hf\":\n            for prm in enc.module.layernorm.parameters():\n                prm.requires_grad = True\n    else:\n        # A CNN exposes no block list to freeze by depth, and its early layers are far\n        # cheaper to relearn than a transformer's, so the whole thing is trained.\n        for prm in enc.parameters():\n            prm.requires_grad = True\n    n_tr = sum(p.numel() for p in enc.parameters() if p.requires_grad)\n    where = (f\"last {min(CFG.unfreeze_last, len(blocks))} of {len(blocks)} blocks\"\n             if blocks else \"all layers (no block structure to freeze by depth)\")\n    log(f\"  trainable: {where}, {n_tr / 1e6:.1f}M params\")\n    return enc","metadata":{"execution":{"iopub.status.busy":"2026-08-07T08:19:45.51401Z","iopub.execute_input":"2026-08-07T08:19:45.514669Z","iopub.status.idle":"2026-08-07T08:19:45.546659Z","shell.execute_reply.started":"2026-08-07T08:19:45.514638Z","shell.execute_reply":"2026-08-07T08:19:45.545631Z"},"papermill":{"duration":0.022715,"end_time":"2026-08-07T04:07:55.276663+00:00","exception":false,"start_time":"2026-08-07T04:07:55.253948+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def augment(imgs):\n    \"\"\"Enhanced multi-scale affine + gamma + noise + cutout augmentation on PyTorch tensors.\"\"\"\n    b = imgs.shape[0]\n    dev = imgs.device\n    x = imgs.float()\n    lead = x.shape[:-3]\n    x = x.reshape(-1, *x.shape[-3:])\n\n    # 1. Random Affine (Rotation +/-10deg, Scale +/-8%, Shift +/-6%, Shear +/-5deg)\n    ang = (torch.rand(b, device=dev) - 0.5) * (2 * 10 * np.pi / 180)   # +/- 10 degrees\n    scale = 1.0 + (torch.rand(b, device=dev) - 0.5) * 0.16             # +/- 8%\n    tx = (torch.rand(b, device=dev) - 0.5) * 0.12                      # +/- 6%\n    ty = (torch.rand(b, device=dev) - 0.5) * 0.12\n    shear_x = (torch.rand(b, device=dev) - 0.5) * (2 * 5 * np.pi / 180) # +/- 5 degrees\n    shear_y = (torch.rand(b, device=dev) - 0.5) * (2 * 5 * np.pi / 180)\n\n    cos, sin = torch.cos(ang), torch.sin(ang)\n    tan_sx, tan_sy = torch.tan(shear_x), torch.tan(shear_y)\n\n    a00 = (cos + sin * tan_sy) / scale\n    a01 = (-sin + cos * tan_sx) / scale\n    a10 = (sin + cos * tan_sy) / scale\n    a11 = (cos - sin * tan_sx) / scale\n\n    theta = torch.stack([torch.stack([a00, a01, tx], 1),\n                         torch.stack([a10, a11, ty], 1)], 1)           # [b,2,3]\n\n    rep = x.shape[0] // b\n    theta = theta.repeat_interleave(rep, 0)\n    grid = F.affine_grid(theta, x.shape, align_corners=False)\n    x = F.grid_sample(x, grid, mode=\"bilinear\", padding_mode=\"zeros\",\n                      align_corners=False)\n\n    # 2. Random Gain (+/- 20%) & Gamma Contrast Variation (gamma in [0.81, 1.22])\n    gain = 1.0 + (torch.rand(b, 1, 1, 1, device=dev) - 0.5) * 0.2\n    x = x * gain.repeat_interleave(rep, 0)\n\n    gamma = torch.exp((torch.rand(b, 1, 1, 1, device=dev) - 0.5) * 0.4)\n    x_norm = (x / 255.0).clamp(1e-4, 1.0)\n    x = torch.pow(x_norm, gamma.repeat_interleave(rep, 0)) * 255.0\n\n    # 3. Gaussian Noise Injection (sigma up to 3.0 on 0-255 scale)\n    if torch.rand(1).item() < 0.5:\n        noise = torch.randn_like(x) * 3.0\n        x = x + noise\n\n    # 4. Random Erasing / Cutout (mild rectangular occlusion)\n    if torch.rand(1).item() < 0.3:\n        _, _, h, w = x.shape\n        ew, eh = int(w * 0.12), int(h * 0.12)\n        y0 = torch.randint(0, max(1, h - eh), (b,), device=dev)\n        x0 = torch.randint(0, max(1, w - ew), (b,), device=dev)\n        for i in range(b):\n            idx_range = slice(i * rep, (i + 1) * rep)\n            x[idx_range, :, y0[i]:y0[i]+eh, x0[i]:x0[i]+ew] = 0.0\n\n    return x.clamp(0, 255).reshape(*lead, *x.shape[-3:]).to(imgs.dtype)\n","metadata":{"execution":{"iopub.status.busy":"2026-08-07T08:19:45.548223Z","iopub.execute_input":"2026-08-07T08:19:45.549352Z","iopub.status.idle":"2026-08-07T08:19:45.574321Z","shell.execute_reply.started":"2026-08-07T08:19:45.54928Z","shell.execute_reply":"2026-08-07T08:19:45.573367Z"},"papermill":{"duration":0.013552,"end_time":"2026-08-07T04:07:55.294806+00:00","exception":false,"start_time":"2026-08-07T04:07:55.281254+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def protocol_features(hdr, studies):\n    \"\"\"Study-level acquisition features from the header pass.\n\n    These are available at test time - `test_series.csv` and the test DICOM headers both\n    exist at inference - so unlike the reports this is a legal feature, not a teacher.\n    It carries real signal because protocol tracks indication: a knee scanned with an\n    extra fat-suppressed axial series was scanned by someone looking for something.\n    \"\"\"\n    if hdr.empty:\n        return pd.DataFrame(index=studies).astype(np.float32)\n    h = hdr.copy()\n    h[\"slot\"] = h[\"plane\"].astype(str).str[0] + \"_\" + h[\"fluid\"].astype(str) + \\\n                \"_\" + h[\"fatsat\"].astype(str)\n    f = pd.crosstab(h[\"StudyInstanceUID\"], h[\"slot\"])\n    agg = h.groupby(\"StudyInstanceUID\").agg(\n        n_series=(\"SeriesInstanceUID\", \"size\"),\n        n_planes=(\"plane\", \"nunique\"),\n        n_fatsat=(\"fatsat\", \"sum\"),\n        n_fluid=(\"fluid\", \"sum\"),\n        med_slices=(\"n_slices\", \"median\"),\n        max_slices=(\"n_slices\", \"max\"),\n        med_px=(\"px\", \"median\"),\n    )\n    f = f.join(agg)\n    return f.reindex(studies).fillna(0.0).astype(np.float32)\n\n\ndef protocol_model(feat_tr, y_tr, groups, feat_te, n_folds=5):\n    \"\"\"Ridge per target, out-of-fold on the training studies, mean over folds at test.\"\"\"\n    from sklearn.linear_model import Ridge\n    from sklearn.pipeline import make_pipeline\n    from sklearn.preprocessing import StandardScaler\n\n    cols = sorted(set(feat_tr.columns) & set(feat_te.columns))\n    # A slot column present in no training study is a constant, and a constant column\n    # makes the normal equations singular rather than merely uninformative.\n    cols = [c for c in cols if feat_tr[c].std() > 1e-8]\n    if not cols:\n        return np.zeros_like(y_tr), np.full((len(feat_te), y_tr.shape[1]), 0.5, np.float32)\n    X, Xt = feat_tr[cols].values, feat_te[cols].values\n    oof = np.zeros_like(y_tr)\n    pred = np.zeros((len(Xt), y_tr.shape[1]), np.float32)\n    for f in range(n_folds):\n        va = np.flatnonzero(groups == f)\n        tr = np.flatnonzero(groups != f)\n        if len(va) == 0 or len(tr) < 10:\n            continue\n        # SVD, not the default Cholesky: the slot-count columns sum exactly to\n        # `n_series`, so the design is perfectly collinear by construction and the\n        # normal equations are singular however large alpha is.\n        m = make_pipeline(StandardScaler(), Ridge(alpha=8.0, solver=\"svd\"))\n        m.fit(X[tr], y_tr[tr])\n        oof[va] = m.predict(X[va])\n        pred += m.predict(Xt) / n_folds\n    return oof, pred","metadata":{"execution":{"iopub.status.busy":"2026-08-07T08:19:45.575788Z","iopub.execute_input":"2026-08-07T08:19:45.576193Z","iopub.status.idle":"2026-08-07T08:19:45.604802Z","shell.execute_reply.started":"2026-08-07T08:19:45.576162Z","shell.execute_reply":"2026-08-07T08:19:45.603438Z"},"papermill":{"duration":0.019132,"end_time":"2026-08-07T04:07:55.319972+00:00","exception":false,"start_time":"2026-08-07T04:07:55.30084+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def select_device():\n    \"\"\"Pick a device, and prove it can run a kernel before committing the run to it.\n\n    `torch.cuda.is_available()` reports only that a driver and a device exist, not that\n    the installed build has code for it. Kaggle's P100 is compute capability 6.0 and the\n    pinned PyTorch ships kernels for sm_70 and up, so availability returns True and the\n    first real launch dies with `no kernel image is available for execution on the\n    device`.\n\n    Left unchecked that happens inside the first training step - which is after the\n    header pass and the twenty-minute cache decode, so the run burns most of an hour of\n    GPU quota to discover something knowable in five seconds. Hence a real launch here,\n    at the top of the run, in the same autocast path training uses.\n    \"\"\"\n    if not torch.cuda.is_available():\n        log(\"device: cpu (no CUDA device reported)\")\n        return torch.device(\"cpu\")\n\n    name = torch.cuda.get_device_name(0)\n    major, minor = torch.cuda.get_device_capability(0)\n    try:\n        x = torch.zeros(8, 3, 32, 32, device=\"cuda\")\n        w = torch.zeros(4, 3, 3, 3, device=\"cuda\")\n        with torch.autocast(\"cuda\"):\n            y = F.conv2d(x, w).flatten(1)\n            y = y @ y.T\n        y.sum().item()\n        torch.cuda.synchronize()\n    except Exception as exc:\n        arches = \" \".join(torch.cuda.get_arch_list())\n        raise RuntimeError(\n            f\"CUDA device '{name}' (sm_{major}{minor}) reports available but cannot \"\n            f\"execute a kernel.\\n  underlying error: {exc}\\n\"\n            f\"  this torch build has kernels for: {arches}\\n\"\n            \"This is the known Kaggle P100 mismatch - the default image's PyTorch \"\n            \"carries no Pascal kernels. Fix: set \\\"machine_shape\\\": \\\"NvidiaTeslaT4\\\" \"\n            \"in kernel-metadata.json, or choose GPU T4 x2 in the notebook's \"\n            \"accelerator settings. Failing here rather than after the cache build.\"\n        ) from exc\n\n    log(f\"device: cuda ({name}, sm_{major}{minor}) - kernel launch verified\")\n    return torch.device(\"cuda\")\n\n\ndef macro_auc(y, p):\n    from sklearn.metrics import roc_auc_score\n    return float(np.nanmean([roc_auc_score(y[:, j], p[:, j])\n                             if len(set(y[:, j].tolist())) > 1 else np.nan\n                             for j in range(y.shape[1])]))\n\n\ndef per_target_auc(y, p):\n    from sklearn.metrics import roc_auc_score\n    return {t: (roc_auc_score(y[:, j], p[:, j])\n                if len(set(y[:, j].tolist())) > 1 else float(\"nan\"))\n            for j, t in enumerate(TARGETS)}\n\n\ndef take_group(rows, g):\n    return rows[:, :, g * CFG.group:(g + 1) * CFG.group]\n\n\n@torch.no_grad()\ndef predict(model, cache, mask, idx, dev, n_group):\n    \"\"\"Average the logits over the groups of each slot.\n\n    Training sees one group at a time, which doubles as augmentation along the stack;\n    inference averages over all of them.\n    \"\"\"\n    model.eval()\n    out = []\n    for b in range(0, len(idx), CFG.eval_batch):\n        sel = idx[b:b + CFG.eval_batch]\n        rows = torch.from_numpy(cache[sel]).to(dev)\n        m = torch.from_numpy(mask[sel]).to(dev)\n        acc = None\n        for g in range(n_group):\n            with torch.autocast(\"cuda\", enabled=dev.type == \"cuda\"):\n                z = model(take_group(rows, g), m).float()\n            acc = z if acc is None else acc + z\n        out.append(torch.sigmoid(acc / n_group).cpu().numpy())\n    return np.concatenate(out) if out else np.zeros((0, len(TARGETS)), np.float32)\n\n\ndef rank_mean(mats, weights=None):\n    \"\"\"Average several prediction matrices in rank space.\n\n    AUC is invariant under any strictly increasing map of a column, so calibration is\n    worth nothing and only order is read. Averaging raw probabilities lets whichever\n    model happens to be most confident dominate; averaging ranks combines the only\n    information the metric reads.\n    \"\"\"\n    mats = [np.asarray(m, np.float64) for m in mats]\n    weights = np.ones(len(mats)) if weights is None else np.asarray(weights, np.float64)\n    weights = weights / weights.sum()\n    acc = np.zeros_like(mats[0])\n    for w, m in zip(weights, mats):\n        acc += w * pd.DataFrame(m).rank(pct=True).values\n    return acc","metadata":{"execution":{"iopub.status.busy":"2026-08-07T08:19:45.606152Z","iopub.execute_input":"2026-08-07T08:19:45.606494Z","iopub.status.idle":"2026-08-07T08:19:45.630276Z","shell.execute_reply.started":"2026-08-07T08:19:45.606466Z","shell.execute_reply":"2026-08-07T08:19:45.629109Z"},"papermill":{"duration":0.019734,"end_time":"2026-08-07T04:07:55.348274+00:00","exception":false,"start_time":"2026-08-07T04:07:55.32854+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. Validating without fooling yourself\n\n**Shared reports.** Some reports are byte-identical across studies — a template read for\nan unremarkable knee — and every study in such a group gets the same derived target.\nSplit the group across folds and the model is scored on a target whose source it trained\non. Folds are assigned by a hash of the report text, which keeps duplicates whole.\n\n**Two references, two meanings.** Performance is reported against the derived targets,\nwhich cover every study and measure whether the imaging model learned what the text\nsays; and against the annotated studies, which are far fewer and measure agreement with\na reading of the *images*. The second resembles the test set. The first has enough\npositives per label to tell a real difference from noise.\n\n**The annotated reference has to be held out.** This is where the source notebook leaks:\nit scores the annotated AUC over *every* gold study, including the ones in the training\nsplit, and then uses that number to choose the checkpoint. Here gold studies take their\nfold assignment like everything else and only held-out gold is scored.\n\n**Choosing an epoch.** The obvious rule — best epoch on the larger reference — is wrong,\nbecause the two measure different things and an epoch that gains on one while losing on\nthe other has been shown to be *different*, not better. Ranking by\n\n$$\\text{score}(e) = \\min\\big(\\mathrm{AUC}^\\text{derived}_e,\\ \\mathrm{AUC}^\\text{annot}_e\\big)$$\n\nrefuses that trade, and makes training length a measured quantity rather than a guess.","metadata":{"papermill":{"duration":0.004593,"end_time":"2026-08-07T04:07:55.357918+00:00","exception":false,"start_time":"2026-08-07T04:07:55.353325+00:00","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def train_fold(cache, mask, Y, W, tr, va, gold_idx, gold_y, n_group, dev, fold):\n    \"\"\"Fit one fold and return (best model, val prediction, chosen-epoch diagnostics).\n\n    Epoch selection reads two references by their *worse* value. They measure different\n    things - the derived targets cover every study and say whether the imaging model\n    learned what the text says, the annotated subset is far smaller and says whether it\n    agrees with a reading of the images - so an epoch that gains on one while losing on\n    the other has not been shown to be better, only different. Taking the minimum\n    refuses that trade, and makes training length a measured quantity rather than a\n    hyperparameter guessed in advance.\n    \"\"\"\n    seed_all(CFG.seed + fold)\n    model = Model(build_encoder()).to(dev)\n    enc_params = [p for p in model.encoder.parameters() if p.requires_grad]\n    groups = [{\"params\": model.head.parameters(), \"lr\": CFG.lr_head}]\n    max_lr = [CFG.lr_head]\n    if enc_params:\n        groups.insert(0, {\"params\": enc_params, \"lr\": CFG.lr_backbone})\n        max_lr.insert(0, CFG.lr_backbone)\n    opt = torch.optim.AdamW(groups, weight_decay=CFG.weight_decay)\n\n    steps = max(CFG.epochs * (len(tr) // CFG.batch_studies), 1)\n    sched = torch.optim.lr_scheduler.OneCycleLR(opt, max_lr=max_lr, total_steps=steps,\n                                                pct_start=0.15)\n    scaler = torch.amp.GradScaler(\"cuda\", enabled=dev.type == \"cuda\")\n\n    yv = (Y[va] > 0.5).astype(int)\n    best, best_state, best_ep = -1.0, None, -1\n    stepped = 0\n\n    for ep in range(CFG.epochs):\n        model.train()\n        perm = np.random.permutation(tr)\n        tot, nstep = 0.0, 0\n        for b in range(0, len(perm) - CFG.batch_studies + 1, CFG.batch_studies):\n            sel = perm[b:b + CFG.batch_studies]\n            rows = torch.from_numpy(cache[sel]).to(dev)\n            g = int(torch.randint(n_group, (1,)).item())\n            imgs = augment(take_group(rows, g))\n            m = torch.from_numpy(mask[sel]).to(dev)\n            y = torch.from_numpy(Y[sel]).to(dev)\n            w = torch.from_numpy(W[sel]).to(dev)\n            with torch.autocast(\"cuda\", enabled=dev.type == \"cuda\"):\n                loss = (F.binary_cross_entropy_with_logits(\n                    model(imgs, m), y, reduction=\"none\") * w).mean()\n            opt.zero_grad(set_to_none=True)\n            scaler.scale(loss).backward()\n            scaler.step(opt)\n            scaler.update()\n            if stepped < steps:\n                sched.step()\n                stepped += 1\n            tot += loss.item()\n            nstep += 1\n\n        pv = predict(model, cache, mask, va, dev, n_group)\n        d_auc = macro_auc(yv, pv)\n\n        # The annotated reference is evaluated on the gold studies *in this fold's\n        # validation split only*. Scoring it over every gold study - which is what the\n        # source notebook does - measures the model on rows it just trained on, and that\n        # number then chooses the checkpoint.\n        g_auc = float(\"nan\")\n        gv = np.intersect1d(gold_idx, va)\n        if len(gv) >= 8:\n            gy = gold_y[np.searchsorted(gold_idx, gv)]\n            g_auc = macro_auc(gy, predict(model, cache, mask, gv, dev, n_group))\n\n        score = d_auc if not np.isfinite(g_auc) else min(d_auc, g_auc)\n        log(f\"  fold {fold} epoch {ep + 1}/{CFG.epochs}  loss {tot / max(nstep, 1):.4f}\"\n            f\"  derived {d_auc:.4f}  annot {g_auc:.4f}  worse-of-two {score:.4f}\")\n        if score > best:\n            best, best_ep = score, ep\n            best_state = {k: v.detach().cpu().clone()\n                          for k, v in model.state_dict().items()}\n        if time.time() - T0 > CFG.time_budget:\n            log(\"  time budget reached inside fold\")\n            break\n\n    if best_state is not None:\n        model.load_state_dict(best_state)\n    log(f\"  fold {fold}: restored epoch {best_ep + 1} (worse-of-two {best:.4f})\")\n    return model, predict(model, cache, mask, va, dev, n_group), best","metadata":{"execution":{"iopub.status.busy":"2026-08-07T08:19:45.632021Z","iopub.execute_input":"2026-08-07T08:19:45.632455Z","iopub.status.idle":"2026-08-07T08:19:45.662241Z","shell.execute_reply.started":"2026-08-07T08:19:45.632416Z","shell.execute_reply":"2026-08-07T08:19:45.661223Z"},"papermill":{"duration":0.018861,"end_time":"2026-08-07T04:07:55.381487+00:00","exception":false,"start_time":"2026-08-07T04:07:55.362626+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def write_submission(study_ids, pred, test_df, path=\"submission.csv\"):\n    sub = pd.DataFrame(pred, columns=TARGETS)\n    sub.insert(0, \"StudyInstanceUID\", study_ids)\n    sub = test_df[[\"StudyInstanceUID\"]].merge(sub, on=\"StudyInstanceUID\", how=\"left\")\n    sub[TARGETS] = sub[TARGETS].fillna(0.5)\n    sub.to_csv(path, index=False)\n    log(f\"{path} {sub.shape}; nulls {int(sub[TARGETS].isna().sum().sum())}\")\n    return sub\n\n\ndef write_benchmark(root, path=\"submission.csv\"):\n    \"\"\"Write the 0.5 benchmark file immediately.\n\n    A submission that never writes scores nothing at all, which is strictly worse than\n    scoring badly. A try/except covers exceptions, but a kill for memory is a SIGKILL and\n    never reaches it, so a valid file has to exist from the first second.\n    \"\"\"\n    t = pd.read_csv(Path(root) / \"test.csv\")\n    for c in TARGETS:\n        t[c] = 0.5\n    t.to_csv(path, index=False)","metadata":{"execution":{"iopub.status.busy":"2026-08-07T08:19:45.663442Z","iopub.execute_input":"2026-08-07T08:19:45.663841Z","iopub.status.idle":"2026-08-07T08:19:45.689069Z","shell.execute_reply.started":"2026-08-07T08:19:45.663789Z","shell.execute_reply":"2026-08-07T08:19:45.687916Z"},"papermill":{"duration":0.011823,"end_time":"2026-08-07T04:07:55.398486+00:00","exception":false,"start_time":"2026-08-07T04:07:55.386663+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. Run\n\nA valid `submission.csv` is written before anything else and overwritten only once real\npredictions exist: a submission that never writes scores nothing at all, which is\nstrictly worse than scoring badly, and a kill for memory is a SIGKILL that no\n`try`/`except` will catch.\n\nKnobs worth touching, all via environment variables so nothing below needs editing:\n`RSNA_FOLDS` (default 5 — drop to 2 if the run is time-bound), `RSNA_TIME_BUDGET`\n(seconds, default 8 h), `RSNA_BACKBONE` (a timm name, overrides the DINOv2 search),\n`SLOT_SCHEME=public` (ablate the header recovery back to the provided flags).","metadata":{"papermill":{"duration":0.004558,"end_time":"2026-08-07T04:07:55.407656+00:00","exception":false,"start_time":"2026-08-07T04:07:55.403098+00:00","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def main(root=None):\n    root = find_root(root)\n    log(f\"input root: {root}\")\n    write_benchmark(root)\n    seed_all()\n    # Before the header pass and the cache decode, not after: an unusable accelerator\n    # is knowable now and costs most of an hour to discover later.\n    dev = select_device()\n\n    test_df = pd.read_csv(root / \"test.csv\")\n    train_df = pd.read_csv(root / \"train.csv\")\n    train_series = pd.read_csv(root / \"train_series.csv\")\n    test_series = pd.read_csv(root / \"test_series.csv\")\n    log(f\"train {train_df.shape} test {test_df.shape}\")\n\n    both = pd.concat([train_series, test_series])\n    plane_map = dict(zip(both[\"SeriesInstanceUID\"], both[\"Anatomical_Plane\"]))\n\n    log(\"header pass: test\")\n    hte = annotate(walk(root, \"test_series\"))\n    log(f\"  {len(hte)} test series\")\n    log(\"header pass: train\")\n    htr = annotate(walk(root, \"train_series\"))\n    log(f\"  {len(htr)} train series\")\n    for h in (htr, hte):\n        if not h.empty:\n            h[\"plane\"] = h[\"SeriesInstanceUID\"].map(plane_map)\n\n    def lat_of(h):\n        \"\"\"Study -> 'L' / 'R' / None. The tag is present on some studies only and is\n        sometimes an empty string rather than absent, which is not the same as NaN.\"\"\"\n        d = {}\n        if h.empty:\n            return d\n        for st, g in h.groupby(\"StudyInstanceUID\"):\n            v = [str(x).strip().upper() for x in g[\"Laterality\"].dropna()]\n            v = [x[0] for x in v if x and x[0] in (\"L\", \"R\")]\n            d[st] = v[0] if v else None\n        return d\n\n    slots_tr, slots_te = pick_slots(htr, plane_map), pick_slots(hte, plane_map)\n    cov = pd.Series([len(v) for v in slots_tr.values()] or [0]).describe()\n    log(f\"train slots per study: mean {cov['mean']:.2f} min {cov['min']:.0f} \"\n        f\"max {cov['max']:.0f}\")\n\n    n_group = plan_cache(len(train_df))\n    log(f\"cache layout: {n_group} groups x {CFG.group} slices\")\n    st_tr, Ctr, Mtr = build_cache(slots_tr, plane_map, lat_of(htr), \"train\", n_group)\n    st_te, Cte, Mte = build_cache(slots_te, plane_map, lat_of(hte), \"test\", n_group)\n\n    # ---- targets ----------------------------------------------------------- #\n    t_lab = time.time()\n    rules = pd.DataFrame([extract(r) for r in train_df[\"Report\"].fillna(\"\")])\n    rules[\"StudyInstanceUID\"] = train_df[\"StudyInstanceUID\"].values\n    rules = rules.set_index(\"StudyInstanceUID\")\n    log(f\"rule labels for {len(rules)} studies in {time.time() - t_lab:.1f}s\")\n\n    # Folds are assigned by a hash of the report text, so byte-identical reports - a\n    # template read for an unremarkable knee - stay whole inside one fold. Split such a\n    # group and the model is scored on a target whose source it has trained on.\n    rep = train_df.set_index(\"StudyInstanceUID\")[\"Report\"].fillna(\"\")\n    fold_of = {s: int(hashlib.md5(rep.get(s, s).encode()).hexdigest()[:8], 16) % CFG.n_folds\n               for s in rep.index}\n\n    # The distilled text model shares those folds for the same reason.\n    order = list(rules.index)\n    rule_mat = rules[TARGETS].values.astype(np.float32)\n    text_oof, text_corr = distill([rep.get(s, \"\") for s in order], rule_mat,\n                                  np.array([fold_of.get(s, 0) for s in order]),\n                                  n_splits=CFG.n_folds, seed=CFG.seed)\n    # Weight the distilled model per target by how well it recovered the rules out of\n    # fold. A target where it failed contributes nothing rather than noise.\n    w_text = CFG.w_text * np.clip(text_corr, 0.0, 1.0)\n    w_text[text_corr < 0.15] = 0.0\n    log(f\"text-model weight per target: min {w_text.min():.2f} max {w_text.max():.2f}, \"\n        f\"{int((w_text == 0).sum())}/{len(TARGETS)} gated out\")\n    derived = blend(rule_mat, text_oof, w_text=w_text)\n    derived = pd.DataFrame(derived, index=order, columns=TARGETS)\n    conf = rules[[t + \"__conf\" for t in TARGETS]].values.astype(np.float32)\n    conf = pd.DataFrame(conf, index=order, columns=TARGETS)\n\n    gold = train_df.set_index(\"StudyInstanceUID\")[TARGETS]\n    gold = gold[gold.notna().all(axis=1)]\n    log(f\"annotated studies: {len(gold)}\")\n\n    Y = np.zeros((len(st_tr), len(TARGETS)), np.float32)\n    W = np.zeros_like(Y)\n    for i, st in enumerate(st_tr):\n        if st in gold.index:\n            Y[i], W[i] = gold.loc[st].values, CFG.gold_weight\n        elif st in derived.index:\n            Y[i] = derived.loc[st].values\n            # Silence on a finding is weak evidence, and should pull weakly.\n            W[i] = 0.25 + 0.75 * conf.loc[st].values\n    keep = np.where(W.sum(1) > 0)[0]\n    folds = np.array([fold_of.get(s, 0) for s in st_tr])\n    log(f\"supervised {len(keep)} of {len(st_tr)} studies\")\n\n    gold_pos = np.array(sorted(i for i, s in enumerate(st_tr) if s in gold.index))\n    gold_y = (gold.loc[[st_tr[i] for i in gold_pos]].values.astype(int)\n              if len(gold_pos) else np.zeros((0, len(TARGETS)), int))\n\n    # ---- protocol-only model ------------------------------------------------ #\n    feat_tr = protocol_features(htr, st_tr)\n    feat_te = protocol_features(hte, st_te)\n    proto_oof, proto_te = protocol_model(feat_tr, Y, folds, feat_te, CFG.n_folds)\n    proto_auc = macro_auc((Y[keep] > 0.5).astype(int), proto_oof[keep])\n    # Diversity is only worth having if it is diversity around something. Protocol\n    # tracks indication - a knee given an extra fat-suppressed axial series was scanned\n    # by someone looking for something - but that is an empirical claim, so it has to\n    # clear chance out-of-fold before it is allowed into the fusion.\n    w_protocol = CFG.w_protocol if proto_auc > 0.52 else 0.0\n    log(f\"protocol model: {feat_tr.shape[1]} features, derived-OOF macro AUC \"\n        f\"{proto_auc:.4f} -> fusion weight {w_protocol:.2f}\")\n\n    # ---- image model, grouped K-fold ---------------------------------------- #\n    oof = np.full((len(st_tr), len(TARGETS)), np.nan, np.float32)\n    test_preds, fold_scores = [], []\n\n    for f in range(min(CFG.max_folds_to_run, CFG.n_folds)):\n        va = np.array([i for i in keep if folds[i] == f])\n        tr = np.array([i for i in keep if folds[i] != f])\n        if len(va) == 0 or len(tr) < CFG.batch_studies:\n            log(f\"fold {f}: too small, skipped\")\n            continue\n        log(f\"fold {f}: train {len(tr)} / val {len(va)} \"\n            f\"(annotated in val: {len(np.intersect1d(gold_pos, va))})\")\n        model, pv, score = train_fold(Ctr, Mtr, Y, W, tr, va, gold_pos, gold_y,\n                                      n_group, dev, f)\n        oof[va] = pv\n        fold_scores.append(score)\n        test_preds.append(predict(model, Cte, Mte, np.arange(len(st_te)), dev, n_group))\n        del model\n        gc.collect()\n        if dev.type == \"cuda\":\n            torch.cuda.empty_cache()\n        if time.time() - T0 > CFG.time_budget:\n            log(\"time budget reached; stopping fold loop\")\n            break\n\n    if not test_preds:\n        log(\"no fold completed - leaving the benchmark submission in place\")\n        return None\n\n    # ---- honest out-of-fold report ------------------------------------------ #\n    done = np.flatnonzero(np.isfinite(oof).all(1))\n    log(f\"OOF over {len(done)} studies\")\n    log(f\"  vs derived targets : {macro_auc((Y[done] > 0.5).astype(int), oof[done]):.4f}\")\n    gd = np.intersect1d(gold_pos, done)\n    if len(gd) >= 8:\n        gy = gold_y[np.searchsorted(gold_pos, gd)]\n        log(f\"  vs annotated ({len(gd)}) : {macro_auc(gy, oof[gd]):.4f}\")\n        for t, v in per_target_auc(gy, oof[gd]).items():\n            log(f\"      {t:<18} {v:.3f}\")\n\n    # ---- fuse and submit ----------------------------------------------------- #\n    image_te = rank_mean(test_preds)\n    final = (image_te if w_protocol == 0.0\n             else rank_mean([image_te, proto_te],\n                            weights=[1.0 - w_protocol, w_protocol]))\n    sub = write_submission(st_te, final, test_df)\n    log(f\"folds run: {len(test_preds)}, mean worse-of-two {np.mean(fold_scores):.4f}\")\n    return sub\n    log(\"done\")","metadata":{"execution":{"iopub.status.busy":"2026-08-07T08:19:45.690485Z","iopub.execute_input":"2026-08-07T08:19:45.690999Z","iopub.status.idle":"2026-08-07T08:19:45.727022Z","shell.execute_reply.started":"2026-08-07T08:19:45.690968Z","shell.execute_reply":"2026-08-07T08:19:45.725792Z"},"papermill":{"duration":0.029309,"end_time":"2026-08-07T04:07:55.441675+00:00","exception":false,"start_time":"2026-08-07T04:07:55.412366+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport traceback\n\ntry:\n    sub = main()\nexcept Exception:\n    traceback.print_exc()\n    # Fall back to the benchmark file rather than dying: an exception here still leaves\n    # a scoreable submission behind.\n    write_benchmark(find_root())\n    sub = pd.read_csv(\"submission.csv\")\n    print(\"wrote fallback submission.csv\")\nlog(\"done\")\nsub.head(20)","metadata":{"execution":{"iopub.status.busy":"2026-08-07T08:19:45.73004Z","iopub.execute_input":"2026-08-07T08:19:45.730458Z","execution_failed":"2026-08-07T08:21:30.791Z"},"papermill":{"duration":4255.263144,"end_time":"2026-08-07T05:18:50.709812+00:00","exception":false,"start_time":"2026-08-07T04:07:55.446668+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null}]}