{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# === Hardcoded Kaggle paths ===\nfrom pathlib import Path\n\n\n# === CELL 2 ===\nfrom __future__ import annotations\n\nimport re\nimport unicodedata\n\nTARGETS = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\",\n    \"Medial OA\", \"Lateral OA\", \"PF OA\", \"Effusion\",\n    \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\",\n]\n\n\n# Turkish dotted/dotless i must be folded before casefolding, otherwise \"İZLENMEZ\"\n# and \"izlenmez\" diverge. ß and the Croatian/Serbian d-with-stroke likewise.\n_PRE = str.maketrans({\n    \"ı\": \"i\", \"İ\": \"i\", \"I\": \"i\", \"ß\": \"ss\", \"đ\": \"d\", \"Đ\": \"d\",\n    \"ø\": \"o\", \"Ø\": \"o\", \"æ\": \"ae\", \"Æ\": \"ae\",\n})\n\n\ndef normalize(text: str) -> str:\n    \"\"\"Fold case, diacritics and separators; keep Greek and Cyrillic letters.\n\n    NFKD decomposition strips Latin accents and Greek tonos alike (ά -> α), which is what\n    we want: reports are inconsistent about accents. It also maps the MICRO SIGN U+00B5\n    to a real mu, which matters because most Greek reports here use the wrong codepoint.\n    \"\"\"\n    if not isinstance(text, str):\n        return \"\"\n    text = text.translate(_PRE).lower()\n    text = unicodedata.normalize(\"NFKD\", text)\n    text = \"\".join(ch for ch in text if not unicodedata.combining(ch))\n    text = text.replace(\"­\", \"\")                    # soft hyphen\n    text = re.sub(r\"[_\\-/\\\\]+\", \" \", text)\n    text = re.sub(r\"[ \\t]+\", \" \", text)\n    return text\n\n\n_SENT_SPLIT = re.compile(r\"(?<=[.;!?])\\s+|\\n+\")\n\n\ndef clauses(text: str):\n    \"\"\"Split into clauses, then attach `header:` lines to the value that follows.\n\n    A report line reading `Fractures :` followed by `Aucune.` is one statement. Splitting\n    on punctuation alone separates the anatomy from its negation and flips the label.\n    \"\"\"\n    norm = normalize(text)\n    raw = [c.strip() for c in _SENT_SPLIT.split(norm) if c and c.strip()]\n\n    merged = []\n    for i, c in enumerate(raw):\n        # A fragment ending in a colon is a heading for the next fragment. Structured\n        # English reports write long ones - \"lateral compartment (meniscus, collateral\n        # ligament complex, cartilage):\" is eight words - so the cap is generous.\n        #\n        # A merged heading must NOT also stand alone. On its own it carries the anatomy\n        # word with no negation in scope, so `Fractures :` / `Aucune.` asserted a fracture\n        # off the heading while the joined clause correctly read the denial. The joined\n        # clause is a superset of the heading, so nothing is lost by dropping it; a\n        # heading with no value beneath it is not merged and still stands.\n        if c.endswith(\":\") and len(c.split()) <= 14 and i + 1 < len(raw):\n            merged.append(c + \" \" + raw[i + 1])\n        else:\n            merged.append(c)\n    # Comma-separated enumerations inside a long clause hide separate assertions.\n    out = []\n    for c in merged:\n        out.append(c)\n        if len(c.split()) > 25:\n            out.extend(p.strip() for p in c.split(\",\") if len(p.split()) > 2)\n    return out\n\n\ndef _rx(*alts: str) -> re.Pattern:\n    return re.compile(\"|\".join(alts))\n\n\n# === CELL 3 ===\nNEGATION = _rx(\n    # en\n    r\"\\bno\\b\", r\"\\bnot\\b\", r\"\\bwithout\\b\", r\"\\bnegative for\\b\", r\"\\babsence\\b\",\n    r\"\\bno evidence\\b\", r\"\\bunremarkable\\b\", r\"\\bfree of\\b\", r\"\\bnone\\b\", r\"\\bnil\\b\",\n    # es\n    r\"\\bsin\\b\", r\"\\bno hay\\b\", r\"\\bausencia\\b\", r\"\\bausentes?\\b\",\n    # fr\n    r\"\\bpas de\\b\", r\"\\bsans\\b\", r\"\\baucune?\\b\", r\"\\babsence\\b\",\n    # nl\n    r\"\\bgeen\\b\", r\"\\bzonder\\b\", r\"\\bniet\\b\",\n    # de\n    r\"\\bkeine?\\b\", r\"\\bohne\\b\", r\"\\bnicht\\b\",\n    # tr\n    r\"\\byok\\b\", r\"\\byoktur\\b\", r\"izlenmemekte\", r\"saptanmadi\", r\"\\bdegil\\b\",\n    r\"gozlenmemekte\", r\"mevcut degil\", r\"eslik etmiyor\", r\"\\bizlenmedi\\b\",\n    # hr / sr / bs\n    r\"\\bnema\\b\", r\"\\bbez\\b\", r\"\\bnisu\\b\", r\"\\bnije\\b\",\n    # el (accents already stripped)\n    r\"\\bδεν\\b\", r\"\\bχωρις\\b\", r\"ουδεν\",\n    # bg / ru\n    r\"\\bбез\\b\", r\"\\bне\\b\", r\"липсва\", r\"\\bняма\\b\",\n)\n\nNORMALITY = _rx(\n    r\"\\bnormal\", r\"\\bintact\\b\", r\"\\bpreserved\\b\", r\"\\bwithin normal limits\\b\",\n    r\"limites normales\", r\"\\bconservad\", r\"\\bintegr\", r\"\\bnormales\\b\",\n    r\"\\bdoga(l|ll)\\b\", r\"korunmus\", r\"\\bnormaldir\\b\", r\"olagan\",\n    r\"\\buredn\", r\"\\bocuvan\", r\"\\bodrzan\", r\"\\bintakt\",\n    r\"φυσιολογικ\", r\"ακεραι\",\n    r\"unauffallig\", r\"regelrecht\", r\"\\bintakt\\b\",\n    r\"нормал\", r\"запазен\", r\"съхранен\", r\"\\bбез особености\\b\",\n    r\"\\bgaaf\\b\", r\"\\bnormaal\\b\",\n)\n\nUNCERTAIN = _rx(\n    r\"\\bpossible\\b\", r\"\\bprobable\\b\", r\"\\bsuspicious\\b\", r\"\\bsuspected\\b\",\n    r\"cannot (be )?exclude\", r\"\\bmay\\b\", r\"\\bquestionable\\b\", r\"\\bequivocal\\b\",\n    r\"\\bposible\\b\", r\"sin criterios categoricos\", r\"\\bdudos\",\n    r\"\\bmuhtemel\\b\", r\"\\bolasi\\b\", r\"\\bsupheli\\b\", r\"\\bizlenim\",\n    r\"\\bmoguce\\b\", r\"\\bvjerojatno\\b\", r\"\\bsumnja\\b\",\n    r\"πιθαν\", r\"υποπτ\",\n    r\"\\bmoglich\", r\"\\bverdachtig\", r\"\\bfraglich\", r\"\\bV\\.a\\.\\b\",\n    r\"\\bвъзможно\\b\", r\"\\bвероятно\\b\", r\"суспект\",\n    r\"\\bmogelijk\\b\", r\"\\bverdacht\\b\",\n)\n\n# Pathology vocabulary shared by the paired rules.\nTEAR = _rx(\n    r\"\\btear\", r\"\\btorn\\b\", r\"\\brupture\", r\"\\bdisruption\\b\", r\"discontinuit\",\n    r\"\\bavuls\",\n    r\"\\brotura\\b\", r\"\\broturas\\b\", r\"\\bruptura\", r\"\\bdesgarro\", r\"\\broto\\b\",\n    r\"\\bdechirure\", r\"\\bdechire\",\n    r\"\\bscheur\", r\"\\bruptuur\", r\"gescheurd\",\n    r\"riss(bildung|e|es)?\\b\", r\"einriss\", r\"\\bruptur\", r\"zerreiss\", r\"\\blasion\",\n    r\"\\byirtik\", r\"\\byirtig\", r\"\\bkopma\\b\", r\"butunluk kaybi\", r\"\\brupturu\\b\",\n    r\"\\bpuknuce\", r\"\\bruptur\", r\"\\bprekid\\b\", r\"\\bpukotin\",\n    r\"ρηξη\", r\"ρηξις\", r\"ρηγμα\",\n    r\"руптура\", r\"разкъсв\", r\"разрив\", r\"скъсв\",\n)\n\nDEGEN = _rx(\n    r\"degenerat\", r\"\\bmucoid\\b\", r\"\\bmyxoid\\b\", r\"\\bfray\", r\"\\bfissur\",\n    r\"dejeneratif\", r\"\\bmukoid\\b\", r\"degenerativn\", r\"εκφυλ\", r\"дегенерат\",\n    r\"\\bμυξοειδ\", r\"\\bμυξωδ\",\n    r\"\\bmuco ?ide\\b\", r\"aufgefasert\",\n)\n\nINJURY = _rx(\n    r\"\\binjur\", r\"\\bsprain\", r\"\\blesion\", r\"\\blasion\", r\"\\bedema\\b\", r\"\\boedema\\b\",\n    r\"\\bodem\\b\", r\"\\bedem\\b\", r\"\\bοιδημα\", r\"\\bодем\", r\"\\bедем\", r\"\\bstrain\\b\",\n    r\"\\bhigh signal\\b\", r\"\\bsignal alteration\\b\", r\"\\bhiperintens\", r\"\\bhyperintens\",\n    r\"aumento de senal\", r\"alteracion de senal\", r\"cambio de senal\",\n    r\"\\bsignalanhebung\", r\"\\bsignalalteration\", r\"verhoogd signaal\", r\"sinyal artis\",\n    r\"αυξημενο σημα\", r\"повишен сигнал\",\n    r\"\\bthicken\", r\"\\bzadebljanje\\b\", r\"\\bverdikking\\b\", r\"\\bdistenzij\",\n    r\"\\blaksite\\b\", r\"\\blaxity\\b\", r\"\\bpartial\\b\", r\"\\bparcijaln\", r\"\\bparcial\",\n    r\"\\bpartiel\", r\"\\bpartiell\",\n)\n\n\n# === CELL 4 ===\nANAT = {\n    \"ACL\": _rx(\n        r\"anterior cruciate\", r\"\\bacl\\b\",\n        r\"cruzado anterior\", r\"\\blca\\b\",\n        r\"croise anterieur\",\n        r\"voorste kruisband\", r\"\\bvkb\\b\",\n        r\"vorderes kreuzband\", r\"vorderen kreuzband\", r\"vordere kreuzband\",\n        r\"on capraz\", r\"\\bocb\\b\",\n        r\"prednji krizni\", r\"prednjeg krizn\",\n        r\"προσθι[οα][^ ]* χιαστ\", r\"προσθιου χιαστου\", r\"χιαστο[^ ]* συνδεσμ\",\n        # \"χιαστοι και πλαγιοι συνδεσμοι\" separates the adjective from its noun, so the\n        # adjective stem has to stand alone. Greek marks cruciate with it unambiguously.\n        r\"\\bχιαστ\\w*\",\n        r\"предна кръстна\", r\"предната кръстна\",\n        # Plural, unqualified: reports routinely clear both cruciates in one clause\n        # (\"Ligamentos cruzados y colaterales dentro de limites normales\"), so the\n        # plural form has to match without a side qualifier or the whole clause is lost.\n        r\"cruciate ligaments\", r\"ligamentos cruzados\", r\"ligaments croises\",\n        r\"kruisbanden\", r\"kreuzbander\", r\"capraz baglar\", r\"krizn[a-z]* ligament[a-z]*\",\n        r\"χιαστοι συνδεσμ\", r\"χιαστων συνδεσμ\", r\"кръстните връзки\", r\"кръстни връзки\",\n    ),\n    \"MCL\": _rx(\n        r\"medial collateral\", r\"\\bmcl\\b\", r\"tibial collateral\",\n        r\"colateral medial\", r\"colateral interno\", r\"\\blcm\\b\",\n        r\"collateral medial\", r\"collateral interne\",\n        r\"mediale collaterale\", r\"binnenband\", r\"\\b(mediale|laterale) banden\\b\",\n        r\"\\bcollaterale banden\\b\",\n        r\"innenband\", r\"mediales? kollateral\",\n        r\"\\bic yan bag\", r\"medial kollateral\", r\"\\biyb\\b\",\n        r\"medijalni kolateraln\", r\"medijalnog kolateraln\",\n        r\"εσω πλαγι\", r\"εσωτερικο πλαγι\", r\"\\bπλαγι\\w* συνδεσμ\", r\"\\bπλαγιοι\\b\",\n        r\"медиален колатерал\", r\"вътрешна странична\", r\"\\bколатерал\\w*\",\n        # Same plural pattern as the cruciates.\n        # \"Ligamentos cruzados y colaterales\" separates the noun from its adjective, so\n        # the adjective has to stand alone as a cue.\n        r\"\\bcolaterales\\b\", r\"\\bcollateraux\\b\", r\"\\bcollateralen\\b\", r\"\\bkolateralni\\b\",\n        r\"collateral ligaments\", r\"ligamentos colaterales\", r\"ligaments collateraux\",\n        r\"collaterale banden\", r\"kollateralbander\", r\"seitenbander\", r\"yan baglar\",\n        r\"kolateraln[a-z]* ligament[a-z]*\", r\"πλαγιοι συνδεσμ\", r\"πλαγιων συνδεσμ\",\n        r\"колатерални връзки\", r\"страничните връзки\",\n    ),\n    \"Medial Meniscus\": _rx(\n        r\"medial meniscus\", r\"\\bmm\\b(?= tear)\", r\"medial menisc\",\n        r\"menisco medial\", r\"menisco interno\",\n        r\"menisque medial\", r\"menisque interne\",\n        r\"mediale meniscus\", r\"binnenmeniscus\",\n        r\"innenmeniskus\", r\"medialen? meniskus\", r\"innenmeniskushinterhorn\",\n        r\"medyal menisk\", r\"\\bic menisk\",\n        r\"medijalni meniskus\", r\"medijalnog meniskusa\", r\"medijalnom meniskusu\",\n        r\"εσω μηνισκ\", r\"μηνισκ[^ ]* του εσω\", r\"εσω διαμερισμα[^.]{0,40}μηνισκ\",\n        r\"медиалния менискус\", r\"медиален менискус\", r\"вътрешния менискус\",\n    ),\n    \"Lateral Meniscus\": _rx(\n        r\"lateral meniscus\", r\"lateral menisc\",\n        r\"menisco lateral\", r\"menisco externo\",\n        r\"menisque lateral\", r\"menisque externe\",\n        r\"laterale meniscus\", r\"buitenmeniscus\",\n        r\"aussenmeniskus\", r\"lateralen? meniskus\",\n        r\"lateral menisk\", r\"\\bdis menisk\",\n        r\"lateralni meniskus\", r\"lateralnog meniskusa\", r\"lateralnom meniskusu\",\n        r\"εξω μηνισκ\", r\"μηνισκ[^ ]* του εξω\", r\"εξω διαμερισμα[^.]{0,40}μηνισκ\",\n        r\"латералния менискус\", r\"латерален менискус\", r\"външния менискус\",\n    ),\n}\n\n# Osteoarthritis is rarely written as \"osteoarthritis\". It is written as cartilage loss,\n# chondropathy grade, joint space narrowing, or osteophytes - scoped to a compartment.\nOA_EVIDENCE = _rx(\n    r\"osteoarthrit\", r\"\\barthros\", r\"\\bgonarthros\", r\"\\bosteoarthros\",\n    r\"chondropath\", r\"chondromalac\", r\"condropat\", r\"condromalac\",\n    r\"cartilage loss\", r\"cartilage thinning\", r\"chondral (loss|defect|ulcer|thinning)\",\n    r\"osteophyt\", r\"osteofit\", r\"osteofyt\", r\"osteofito\", r\"osteophyten\",\n    r\"joint space narrowing\", r\"pinzamiento articular\",\n    r\"kikirdak kayb\", r\"kikirdak incelme\", r\"kondropati\", r\"kondral\",\n    r\"kraakbeen(lijden|verlies)\", r\"gonartrose\", r\"artrose\",\n    r\"knorpel(verlust|schaden|defekt)\", r\"arthrose\", r\"gonarthrose\",\n    r\"hrskavic\", r\"hondromalac\", r\"artroz\", r\"osteoartrit\",\n    r\"χονδρ[^ ]*παθ\", r\"αρθριτ\", r\"αρθρωσ\", r\"οστεοφυτ\",\n    r\"αρθρικου χονδρου\", r\"εξαλειψη του αρθρικου χονδρου\",\n    r\"артроз\", r\"хондропат\", r\"остеофит\", r\"хрущял[^.]{0,30}(изтън|увред|дефект)\",\n    r\"ulcera[s]? condral\", r\"cartilago[^.]{0,25}(perdida|adelgaz)\",\n    r\"icrs grade\", r\"outerbridge\",\n)\n\nCOMPARTMENT = {\n    \"Medial OA\": _rx(\n        r\"medial (femorotibial|tibiofemoral|compartment)\",\n        r\"compartimento femorotibial medial\", r\"femorotibial interno\",\n        r\"mediaal femorotibiaal\", r\"mediale femorotibial\",\n        r\"medial femorotibial\", r\"medialen kompartiment\", r\"innere[sn]? kompartiment\",\n        r\"medyal femorotibial\", r\"ic kompartman\", r\"medyal kompartman\",\n        r\"medijaln[^ ]* (femorotibi|odjelj|kompartm)\",\n        r\"εσω διαμερισμα\", r\"εσω κνημιαι\", r\"εσω μηριαι\",\n        r\"медиалн[^ ]* (компартм|отдел|тибиал|феморотиб)\",\n        r\"medial (femoral|tibial) (condyle|plateau)\", r\"condilo femoral medial\",\n        r\"medialen? (femurkondyl|tibiaplateau)\", r\"mediale femorale condyl\",\n    ),\n    \"Lateral OA\": _rx(\n        r\"lateral (femorotibial|tibiofemoral|compartment)\",\n        r\"compartimento femorotibial lateral\", r\"femorotibial externo\",\n        r\"lateraal femorotibiaal\", r\"laterale femorotibial\",\n        r\"lateral femorotibial\", r\"lateralen kompartiment\", r\"aussere[sn]? kompartiment\",\n        r\"lateral femorotibial\", r\"dis kompartman\", r\"lateral kompartman\",\n        r\"lateraln[^ ]* (femorotibi|odjelj|kompartm)\",\n        r\"εξω διαμερισμα\", r\"εξω κνημιαι\", r\"εξω μηριαι\",\n        r\"латералн[^ ]* (компартм|отдел|тибиал|феморотиб)\",\n        r\"lateral (femoral|tibial) (condyle|plateau)\", r\"condilo femoral lateral\",\n        r\"lateralen? (femurkondyl|tibiaplateau)\", r\"laterale femorale condyl\",\n    ),\n    \"PF OA\": _rx(\n        r\"patellofemoral\", r\"femoropatellar\", r\"femoropatelar\", r\"patelofemoral\",\n        r\"retropatellar\", r\"retrorotulian\", r\"\\btrochlea\", r\"\\btroclea\", r\"\\btroklea\",\n        r\"\\bpatella\\b\", r\"\\bpatellar\\b\", r\"\\brotulian\", r\"\\brotula\\b\", r\"\\bpatele\\b\",\n        r\"\\bpatellae?\\b\", r\"patellofemoraal\", r\"femoropatellair\",\n        r\"επιγονατιδ\", r\"μηροεπιγονατιδ\", r\"τροχιλ\",\n        r\"пател\", r\"феморопател\", r\"тролх\",\n        r\"anterior compartment\", r\"compartimento anterior\", r\"prednj[^ ]* odjeljk\",\n    ),\n}\n\n# Self-declaring findings: the term itself is the finding.\nDIRECT = {\n    \"Effusion\": _rx(\n        r\"\\beffusion\", r\"joint fluid\", r\"intra ?articular fluid\", r\"\\bhydrops\\b\",\n        r\"derrame articular\", r\"\\bderrame\\b\", r\"liquido articular\",\n        r\"epanchement\",\n        r\"gewrichtsvocht\", r\"\\bvocht\\b\", r\"\\bhydrops\\b\", r\"gewrichtseffusie\",\n        r\"gelenkerguss\", r\"\\berguss\\b\", r\"gelenksergu\",\n        # \"diz eklemi ici sivi miktari ... artmis\" and \"eklem icerisinde yaygin sivi\n        # artisi\" both occur; the noun takes a possessive suffix, so `eklem ` alone\n        # misses. Match the stem plus any suffix.\n        r\"eklem\\w* ic\\w* sivi\", r\"efuzyon\", r\"eklem sivisi\",\n        r\"sivi (miktari|artisi|birikimi)\", r\"sivi artis\", r\"\\bsivi\\b[^.]{0,25}artmis\",\n        r\"\\bizljev\", r\"\\bizliv\", r\"zglobn[^ ]* tekucin\", r\"\\bhidrops\\b\",\n        r\"αρθρικ[^ ]* υγρ\", r\"υγρου ενδαρθρικα\", r\"ενδαρθρικ[^ ]* υγρ\", r\"ποσοτητα υγρου\",\n        r\"ενδαρθρικ\", r\"αρθρικη συλλογη\", r\"υγρο στην αρθρωση\", r\"υγρου στην αρθρωση\",\n        r\"ставен излив\", r\"излив\", r\"ставна течност\", r\"синовиална течност\",\n    ),\n    \"Synovitis\": _rx(\n        r\"synovit\", r\"sinovit\", r\"synovial (thickening|proliferation|hypertroph)\",\n        r\"synovitis\", r\"synoviale? (verdikking|proliferatie)\",\n        r\"synovialitis\", r\"synovialis(verdickung|proliferation)\",\n        r\"sinovijalitis\", r\"sinovitis\", r\"zadebljanje sinovij\",\n        r\"υμενιτιδα\", r\"συνοβιτιδα\", r\"υμενικ[^ ]* υπερτροφ\", r\"αρθρικου υμεν\",\n        r\"синовит\", r\"синовиал[^ ]* (задебел|пролифер)\",\n        r\"verdikkingen van (het )?synovium\", r\"pannus\",\n    ),\n    \"Baker's\": _rx(\n        r\"baker\", r\"popliteal cyst\", r\"quiste popliteo\", r\"quistes popliteos\",\n        r\"kyste poplite\", r\"popliteale? cyst\", r\"poplitealzyste\", r\"bakerzyste\",\n        r\"popliteal kist\", r\"\\bbakerova\\b\", r\"poplitealn[^ ]* cist\",\n        r\"κυστη baker\", r\"πολυχωρη συνοβιακη κυστη\", r\"κυστη του baker\",\n        r\"киста на бейкър\", r\"бейкърова киста\", r\"поплитеална киста\",\n        r\"gastrocnemio ?semimembranos\", r\"gastrocnemius semimembranosus burs\",\n    ),\n    \"Contusion\": _rx(\n        r\"\\bcontusion\", r\"bone bruise\", r\"bone marrow (o?edema|contusion)\",\n        r\"\\bkontuz\", r\"medular bone o?edema\", r\"marrow o?edema\",\n        r\"contusion osea\", r\"edema oseo\", r\"edema de medula osea\",\n        r\"oedeme osseux\", r\"contusion osseuse\",\n        r\"botcontusie\", r\"botoedeem\", r\"beenmergoedeem\", r\"botmergoedeem\",\n        r\"knochenmarkodem\", r\"knochenodem\", r\"kontusion\", r\"bone bruise\",\n        r\"kemik kontuzyonu\", r\"kemik iligi odemi\", r\"kemik odemi\",\n        r\"kostani edem\", r\"edem kosti\", r\"kontuzij\",\n        r\"οστεομυελικ[^ ]* οιδημα\", r\"οστικο οιδημα\", r\"μυελικο οιδημα\",\n        r\"костномозъчен едем\", r\"костен едем\", r\"контузионен\",\n    ),\n    \"Fracture\": _rx(\n        r\"\\bfractur\", r\"\\bfract\\b\",\n        r\"\\bfractura\", r\"\\bfracturas\\b\",\n        r\"\\bfractuur\", r\"\\bbreuk\\b\",\n        r\"\\bfraktur\", r\"\\bbruch\\b\",\n        r\"\\bkirik\\b\", r\"\\bkirigi\\b\", r\"\\bkirik\\b\",\n        r\"\\bfraktur\", r\"\\bprijelom\", r\"impresijsk[^ ]* fraktur\",\n        r\"καταγμα\", r\"καταγματ\",\n        r\"фрактур\", r\"счупван\", r\"фисур\",\n        r\"insufficiency fracture\", r\"stress fracture\", r\"avulsion fracture\",\n        r\"subchondral fracture\", r\"subkondral kiri\",\n    ),\n}\n\n# Terms that look like a finding but are not the finding being scored.\nDECOY = {\n    # `no fracture` is deliberately absent: a decoy skips the clause, so listing it here\n    # turned the commonest English denial into silence, and the study then pulled on the\n    # fracture head with the weight of a report that never mentioned fractures at all.\n    # `microfractur` is a surgical procedure and `fracture risk` a prediction; both stay.\n    \"Fracture\": _rx(r\"microfractur\", r\"\\bfracture (risk|prophyla)\"),\n    \"Baker's\": _rx(r\"meniscal cyst\", r\"quiste meniscal\", r\"ganglion\"),\n}\n\nPAIRED = {\"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\"}\nOA_TARGETS = {\"Medial OA\", \"Lateral OA\", \"PF OA\"}\n\n\n# === CELL 5 ===\nSTEM_MENISCUS = _rx(r\"menisc\\w*\", r\"menisk\\w*\", r\"μηνισκ\\w*\", r\"мениск\\w*\")\nSTEM_CRUCIATE = _rx(r\"cruciate\", r\"cruzado\", r\"croise\", r\"kruisband\", r\"kreuzband\",\n                    r\"capraz bag\\w*\", r\"krizn\\w*\", r\"χιαστ\\w*\", r\"кръстн\\w*\",\n                    r\"\\bacl\\b\", r\"\\bpcl\\b\", r\"\\blca\\b\", r\"\\blcp\\b\", r\"\\bvkb\\b\",\n                    r\"\\bhkb\\b\", r\"\\bocb\\b\", r\"\\bacb\\b\")\nSTEM_COLLATERAL = _rx(r\"collateral\\w*\", r\"colateral\\w*\", r\"kollateral\\w*\",\n                      r\"collaterale\\w*\", r\"kolateraln\\w*\", r\"yan bag\\w*\",\n                      r\"πλαγι\\w*\", r\"колатерал\\w*\", r\"странич\\w*\",\n                      r\"innenband\\w*\", r\"aussenband\\w*\", r\"binnenband\\w*\",\n                      r\"\\bmcl\\b\", r\"\\blcl\\b\", r\"\\blcm\\b\", r\"\\biyb\\b\")\n\nSIDE_MEDIAL = _rx(r\"\\bmedial\\w*\", r\"\\bmedyal\\w*\", r\"\\bmedijaln\\w*\", r\"\\bmediaal\\w*\",\n                  r\"\\bmediale\\w*\", r\"\\bintern[oa]\\w*\", r\"\\binterne\\w*\", r\"\\binnen\\w*\",\n                  r\"\\bic\\b\", r\"\\bunutarnj\\w*\", r\"\\bεσω\\w*\", r\"\\bεσωτερικ\\w*\",\n                  r\"\\bмедиал\\w*\", r\"\\bвътреш\\w*\", r\"\\btibial collateral\\b\",\n                  r\"\\bbinnen\\w*\", r\"\\bmediaal\\b\")\nSIDE_LATERAL = _rx(r\"\\blateral\\w*\", r\"\\bextern[oa]\\w*\", r\"\\bexterne\\w*\", r\"\\bdis\\b\",\n                   r\"\\blateraln\\w*\", r\"\\baussen\\w*\", r\"\\bbuiten\\w*\", r\"\\bεξω\\w*\",\n                   r\"\\bεξωτερικ\\w*\", r\"\\bлатерал\\w*\", r\"\\bвъншн\\w*\",\n                   r\"\\bfibular collateral\\b\", r\"\\bvanjsk\\w*\")\nSIDE_ANTERIOR = _rx(r\"\\banterior\\w*\", r\"\\bant\\b\", r\"\\bon\\b\", r\"\\bprednj\\w*\",\n                    r\"\\bvorder\\w*\", r\"\\bvoorste\\b\", r\"\\bπροσθι\\w*\", r\"\\bпредн\\w*\",\n                    r\"\\banteriyor\\w*\", r\"\\bavant\\b\", r\"\\bant[eé]rieur\\w*\")\n\n# The contrary of SIDE_ANTERIOR, needed only to stop a side-blind cruciate cue firing on\n# the posterior ligament. It is never used to assert a target - there is no PCL target -\n# so it is deliberately narrow: `posterior horn` is one of the commonest phrases in a\n# knee report and must not be read as a cruciate qualifier, which is why the guard below\n# tests proximity to the cruciate stem rather than presence in the clause.\nSIDE_POSTERIOR = _rx(r\"\\bposterior\\w*\", r\"\\bpost[eé]rieur\\w*\", r\"\\bposteriore\\w*\",\n                     r\"\\bhinter\\w*\", r\"\\bachterste\\b\", r\"\\barka\\b\", r\"\\bstraznj\\w*\",\n                     r\"\\bzadnj\\w*\", r\"\\bοπισθι\\w*\", r\"\\bзадн\\w*\", r\"\\bpostero\\w*\")\n# Fracture is the target whose stem varies most across the corpus.\nSTEM_FRACTURE = _rx(r\"fractur\\w*\", r\"fraktur\\w*\", r\"fractuur\\w*\", r\"\\bfract\\b\",\n                    r\"kiri[kgğ]\\w*\", r\"prijelom\\w*\", r\"lom kosti\", r\"\\bbreuk\\w*\",\n                    r\"\\bbruch\\w*\", r\"καταγμα\\w*\", r\"καταγματ\\w*\", r\"фрактур\\w*\",\n                    # NOT a bare `fissur\\w*`: \"fisuras condrales\" and \"full thickness\n                    # fissures in the articular cartilage\" describe cartilage, not bone.\n                    # The stem has to be anchored to a bone word to mean a fracture.\n                    r\"счупван\\w*\", r\"fisur\\w* (osea|oseas|kost)\", r\"fissur\\w* kost\")\n\nSTEM_OA_COMPARTMENT = _rx(r\"compartment\\w*\", r\"compartimento\\w*\", r\"compartiment\\w*\",\n                          r\"kompartman\\w*\", r\"kompartiment\\w*\", r\"odjelj\\w*\",\n                          r\"διαμερισμα\\w*\", r\"компартм\\w*\", r\"\\bотдел\\w*\",\n                          r\"femorotibial\\w*\", r\"femorotibiaal\\w*\", r\"tibiofemoral\\w*\",\n                          r\"femoro tibial\\w*\", r\"κνημιαι\\w*\", r\"μηριαι\\w*\",\n                          r\"femoral condyl\\w*\", r\"tibial plateau\\w*\",\n                          r\"condilo femoral\", r\"platillo tibial\", r\"tibiaplateau\\w*\",\n                          r\"femurkondyl\\w*\", r\"femoralne? kondil\\w*\",\n                          r\"tibijaln\\w* plato\", r\"femoral kondil\\w*\",\n                          r\"tibia plato\", r\"tibyal plato\")\n\n\ndef _distance(clause: str, stem_rx: re.Pattern, qual_rx: re.Pattern, window: int = 55):\n    \"\"\"Characters from the nearest stem to the nearest qualifier, or None if none is near.\n\n    Character windows rather than token windows, because word order differs: English\n    puts the side before the noun, Greek and Bulgarian often after, and Turkish\n    attaches it as a separate preceding adjective.\n\n    A distance rather than a yes. Presence is enough to decide that a qualifier applies\n    to a structure, but not enough to decide which of two qualifiers applies: a knee\n    report says \"anterior horn\" and \"posterior horn\" constantly, so any window wide\n    enough to catch a real side word also catches an unrelated one, and two rules that\n    both answer \"yes\" cannot be told apart. Comparing how far away they are can.\n    \"\"\"\n    best = None\n    for m in stem_rx.finditer(clause):\n        lo = max(0, m.start() - window)\n        hi = min(len(clause), m.end() + window)\n        for q in qual_rx.finditer(clause[lo:hi]):\n            qs, qe = lo + q.start(), lo + q.end()\n            d = 0 if qs < m.end() and qe > m.start() else \\\n                min(abs(m.start() - qe), abs(qs - m.end()))\n            best = d if best is None else min(best, d)\n    return best\n\n\ndef _near(clause: str, stem_rx: re.Pattern, qual_rx: re.Pattern, window: int = 55):\n    \"\"\"True if a stem match has a qualifier within `window` characters either side.\"\"\"\n    return _distance(clause, stem_rx, qual_rx, window) is not None\n    return False\n\n\n# concept -> (stem, side) pairs used in addition to the phrase lexicons above\nSTEM_RULES = {\n    \"ACL\": (STEM_CRUCIATE, SIDE_ANTERIOR),\n    \"MCL\": (STEM_COLLATERAL, SIDE_MEDIAL),\n    \"Medial Meniscus\": (STEM_MENISCUS, SIDE_MEDIAL),\n    \"Lateral Meniscus\": (STEM_MENISCUS, SIDE_LATERAL),\n    \"Medial OA\": (STEM_OA_COMPARTMENT, SIDE_MEDIAL),\n    \"Lateral OA\": (STEM_OA_COMPARTMENT, SIDE_LATERAL),\n}\n\n\n# === CELL 6 ===\nSEV_LOW = _rx(\n    r\"\\bsmall\\b\", r\"\\bminimal\\b\", r\"\\btrace\\b\", r\"\\bmild\\b\", r\"\\bslight\\b\",\n    r\"\\btiny\\b\", r\"\\bscant\\b\", r\"\\bmimimal\\b\", r\"\\bdiscrete\\b\", r\"\\bfocal\\b\",\n    r\"\\bleve\\b\", r\"\\bminim\", r\"\\bpeque\", r\"\\bligero\\b\", r\"\\bescaso\\b\", r\"\\bdiscreto\\b\",\n    r\"\\bhafif\\b\", r\"\\bminimal\\b\", r\"\\baz miktarda\\b\", r\"\\bsilik\\b\",\n    r\"\\bmanja\\b\", r\"\\bmanji\\b\", r\"\\bblago\\b\", r\"\\bdiskretn\", r\"\\bmalo\\b\",\n    r\"\\bgering\", r\"\\bdiskret\", r\"\\bkleine?r?\\b\", r\"\\bwenig\\b\", r\"\\bzarte?\\b\",\n    r\"\\bbeperkte?\\b\", r\"\\bgeringe\\b\", r\"\\bweinig\\b\", r\"\\blichte?\\b\",\n    r\"\\bηπι\", r\"\\bμικρ\", r\"\\bελαχιστ\",\n    r\"\\bминимал\", r\"\\bлек\", r\"\\bмалк\", r\"\\bнеголям\",\n)\n\nSEV_HIGH = _rx(\n    r\"\\blarge\\b\", r\"\\bmarked\\b\", r\"\\bmassive\\b\", r\"\\bsevere\\b\", r\"\\bextensive\\b\",\n    r\"\\bmoderate\\b\", r\"\\bgross\\b\", r\"\\bsignificant\\b\", r\"\\babundant\\b\", r\"\\btense\\b\",\n    r\"\\bmoderad\", r\"\\bimportante\\b\", r\"\\bsevera?\\b\", r\"\\bmarcad\", r\"\\bcuantios\",\n    r\"\\bbelirgin\\b\", r\"\\byaygin\\b\", r\"\\bileri\\b\", r\"\\bciddi\\b\", r\"\\bbol\\b\",\n    r\"\\bopsezan\\b\", r\"\\bveliki\\b\", r\"\\bizrazit\", r\"\\bznacajn\", r\"\\bumjeren\",\n    r\"\\bausgepragt\", r\"\\bdeutlich\", r\"\\bmassiv\", r\"\\bmassig\", r\"\\bgross\",\n    r\"\\buitgebreid\", r\"\\bgevorderd\", r\"\\bveel\\b\", r\"\\bmatige?\\b\",\n    r\"\\bμετρι\", r\"\\bμεγαλ\", r\"\\bεκτεταμεν\", r\"\\bευμεγεθ\", r\"\\bσοβαρ\",\n    r\"\\bголям\", r\"\\bизразен\", r\"\\bзначим\", r\"\\bумерен\", r\"\\bобилен\",\n)\n\n# OA is often asserted for the whole joint rather than per compartment\n# (\"tricompartmental osteoarthritis\", \"gonarthrose\", \"incipient OA of all three\n# compartments\"). Those statements are evidence for all three OA targets.\nGLOBAL_OA = _rx(\n    r\"tri ?compartment\", r\"all three compartment\", r\"global(ised)? (oa|osteoarthrit)\",\n    r\"\\bgonarthros\", r\"\\bgonartros\", r\"\\bgonarthrose\", r\"\\bgonartrose\",\n    r\"osteoarthritis of the knee\", r\"artrosis (de |)(la )?rodilla\", r\"knee osteoarthrit\",\n    r\"\\bdiz osteoartrit\", r\"\\bgonartroz\", r\"artroza koljena\",\n    r\"οστεοαρθριτιδα\", r\"αρθριτιδα του γονατος\",\n    r\"артроза на колянната\", r\"гонартроз\",\n    r\"degenerative joint disease\", r\"\\bdjd\\b\",\n)\n\n# A bare \"bone marrow oedema\" is not a contusion when it sits under a cartilage\n# defect: subchondral oedema beneath a worn compartment is reactive degenerative signal,\n# and reading it as a bruise turns every osteoarthritic knee into a trauma case.\nDEGENERATIVE_MARROW = _rx(\n    r\"subchondral\", r\"subcondral\", r\"subkondral\", r\"supkondraln\", r\"subchondraln\",\n    r\"υποχονδρι\", r\"субхондрал\", r\"subchondrale?\",\n    r\"\\bcyst\", r\"\\bquist\", r\"\\bzyste\\b\", r\"\\bcistic\", r\"reactive\", r\"reactivo\",\n)\n\nTRAUMA = _rx(\n    r\"\\bbruise\\b\", r\"\\bcontusion\", r\"\\bkontuz\", r\"\\bcontusion osea\\b\",\n    r\"\\btrauma\", r\"\\bimpaction\\b\", r\"\\bpivot shift\\b\", r\"\\bkissing\\b\",\n    r\"\\bacute\\b\", r\"\\bagudo\\b\", r\"\\bakut\", r\"\\bpivot kaymasi\\b\",\n    r\"\\bcontusion osseuse\\b\", r\"\\bbone bruise\\b\", r\"\\bbotcontusie\\b\",\n    r\"\\bконтузион\", r\"\\bμωλωπ\", r\"\\bkontuzij\",\n)\n\n\n# === CELL 7 ===\ndef _polarity(clause: str, anchor_end: int) -> str:\n    \"\"\"Classify one clause as positive, negative or uncertain for a matched term.\n\n    Scope is the whole clause. Clause segmentation already keeps statements short, and\n    a window in characters mis-scopes badly across languages with different word orders -\n    Turkish puts its negator at the end of the sentence, English at the front.\n    \"\"\"\n    if UNCERTAIN.search(clause):\n        return \"uncertain\"\n    if NEGATION.search(clause):\n        return \"negative\"\n    if NORMALITY.search(clause):\n        # \"meniscus normal\" negates; \"normal ... but tear\" does not.\n        if TEAR.search(clause) or re.search(r\"\\bgrade [34]\\b\", clause):\n            return \"positive\"\n        return \"negative\"\n    return \"positive\"\n\n\nclass _Matcher:\n    \"\"\"Phrase lexicon first, stem+side proximity as the fallback.\n\n    Exposes `.search` so it drops into the same slot as a compiled pattern.\n    \"\"\"\n\n    def __init__(self, phrase_rx, stem=None, side=None, window=55, contrary=None):\n        self.phrase_rx = phrase_rx\n        self.stem = stem\n        self.side = side\n        self.window = window\n        self.contrary = contrary\n\n    def search(self, clause):\n        m = self.phrase_rx.search(clause)\n        if m is not None and not self._wrong_side(clause):\n            return m\n        if self.stem is not None and _near(clause, self.stem, self.side, self.window):\n            return self.stem.search(clause)\n        return None\n\n    def _wrong_side(self, clause):\n        \"\"\"True when the clause names the other member of this structure's pair.\n\n        Some cues in the lexicon are side-blind by design: Greek separates the adjective\n        from its noun (\"cruciate and collateral ligaments\"), so the bare adjective stem\n        has to stand alone or the clause is lost. That stem then also matches the\n        posterior cruciate and the lateral collateral, neither of which is a target here,\n        and a positive outranks every negative in the scorer - so one PCL clause was\n        enough to override an explicit \"the ACL is normal\".\n\n        The test is proximity to the structure's own stem, not presence in the clause.\n        \"Posterior horn of the medial meniscus\" appears in a large share of knee reports\n        and says nothing about a cruciate; only a qualifier sitting beside the ligament\n        word is one. A clause naming both sides keeps the match, because it does mention\n        this target.\n        \"\"\"\n        if self.contrary is None or self.stem is None:\n            return False\n        other = _distance(clause, self.stem, self.contrary, self.window)\n        if other is None:\n            return False\n        own = _distance(clause, self.stem, self.side, self.window)\n        # Not \"is the other side mentioned\" but \"is it the nearer of the two\". A clause\n        # reading \"tear of the posterior horn of the medial meniscus; the cruciate\n        # ligaments are intact\" mentions posterior, and under a presence test that was\n        # enough to suppress the cruciate cue - which is the opposite of the intent,\n        # since the clause does describe the ligaments. Ties go to keeping the match:\n        # a clause naming both sides does mention this one.\n        return own is None or other < own\n\n\n# Which cue, if it sits beside the structure's stem, means the clause is about the other\n# member of the pair. Only the two structures with a side-blind cue need one.\nCONTRARY = {\"ACL\": SIDE_POSTERIOR, \"MCL\": SIDE_LATERAL}\n\nANAT_MATCH = {\n    tgt: _Matcher(ANAT[tgt], *STEM_RULES[tgt], contrary=CONTRARY.get(tgt))\n    for tgt in PAIRED\n}\nCOMPARTMENT_MATCH = {\n    \"Medial OA\": _Matcher(COMPARTMENT[\"Medial OA\"], *STEM_RULES[\"Medial OA\"]),\n    \"Lateral OA\": _Matcher(COMPARTMENT[\"Lateral OA\"], *STEM_RULES[\"Lateral OA\"]),\n    \"PF OA\": _Matcher(COMPARTMENT[\"PF OA\"]),\n}\nDIRECT_MATCH = {\n    tgt: _Matcher(_rx(rx.pattern, STEM_FRACTURE.pattern) if tgt == \"Fracture\" else rx)\n    for tgt, rx in DIRECT.items()\n}\n\n\ndef _severity(clause: str) -> float:\n    \"\"\"Weight one positive mention by how emphatic the sentence is.\n\n    Ordered, not calibrated. A \"moderate effusion\" must outrank a \"trace effusion\" and\n    both must outrank silence; the absolute numbers do not matter to AUC.\n    \"\"\"\n    high = SEV_HIGH.search(clause) is not None\n    low = SEV_LOW.search(clause) is not None\n    if high and not low:\n        return 1.0\n    if low and not high:\n        return 0.45\n    return 0.75                       # unqualified mention\n\n\ndef _score_clauses(cls, anat_rx, path_rx=None, decoy_rx=None, context_penalty=None,\n                   context_bonus=None):\n    \"\"\"Accumulate graded evidence over clauses for one target.\n\n    Returns (score, confidence, n_pos, n_neg). Positives are graded by severity and by\n    optional context regexes; negatives only matter when nothing positive was found,\n    because reports assert normality for every structure they check.\n    \"\"\"\n    n_pos = n_neg = n_unc = 0\n    best = 0.0\n    for c in cls:\n        m = anat_rx.search(c)\n        if not m:\n            continue\n        if decoy_rx is not None and decoy_rx.search(c):\n            continue\n        if path_rx is not None and not path_rx.search(c):\n            if NORMALITY.search(c) and not NEGATION.search(c):\n                n_neg += 1\n            continue\n        pol = _polarity(c, m.end())\n        if pol == \"positive\":\n            n_pos += 1\n            w = _severity(c)\n            if context_penalty is not None and context_penalty.search(c):\n                w *= 0.45\n            if context_bonus is not None and context_bonus.search(c):\n                w = min(1.0, w * 1.35)\n            best = max(best, w)\n        elif pol == \"negative\":\n            n_neg += 1\n        else:\n            n_unc += 1\n            best = max(best, 0.30)\n\n    if n_pos or n_unc:\n        # 0.52 .. 0.95, ordered by the strongest single mention, nudged by repetition.\n        score = min(0.95, 0.50 + 0.42 * best + 0.03 * min(n_pos, 3))\n        conf = min(1.0, 0.55 + 0.15 * n_pos)\n    elif n_neg:\n        score = max(0.04, 0.20 - 0.04 * n_neg)\n        conf = min(0.9, 0.45 + 0.12 * n_neg)\n    else:\n        score, conf = 0.28, 0.05          # silence sits above asserted-negative\n    return score, conf, n_pos, n_neg\n\n\ndef extract(report: str) -> dict:\n    \"\"\"Extract twelve (score, confidence) pairs from one report.\"\"\"\n    cls = clauses(report)\n    out = {}\n    path_paired = _rx(TEAR.pattern, DEGEN.pattern, INJURY.pattern)\n\n    for tgt in TARGETS:\n        if tgt in PAIRED:\n            s, c, npos, nneg = _score_clauses(cls, ANAT_MATCH[tgt], path_paired)\n        elif tgt in OA_TARGETS:\n            s, c, npos, nneg = _score_clauses(cls, COMPARTMENT_MATCH[tgt], OA_EVIDENCE)\n        elif tgt == \"Contusion\":\n            # Reactive subchondral oedema under a cartilage defect is osteoarthritis,\n            # not a bruise. Explicit trauma wording pushes the other way.\n            s, c, npos, nneg = _score_clauses(cls, DIRECT_MATCH[tgt], None, DECOY.get(tgt),\n                                              context_penalty=DEGENERATIVE_MARROW,\n                                              context_bonus=TRAUMA)\n        else:\n            s, c, npos, nneg = _score_clauses(cls, DIRECT_MATCH[tgt], None, DECOY.get(tgt))\n        out[tgt] = s\n        out[tgt + \"__conf\"] = c\n        out[tgt + \"__npos\"] = npos\n        out[tgt + \"__nneg\"] = nneg\n\n    # --- cross-target corrections ------------------------------------------ #\n    # A whole-joint osteoarthritis statement is evidence for every compartment that was\n    # not separately assessed. Without this, \"incipient OA of all three compartments\"\n    # scores zero on all three OA targets.\n    g_hits = [c for c in cls if GLOBAL_OA.search(c) and _polarity(c, 0) == \"positive\"]\n    if g_hits:\n        gscore = 0.50 + 0.42 * max(_severity(c) for c in g_hits)\n        for tgt in OA_TARGETS:\n            if out[tgt + \"__npos\"] == 0 and out[tgt + \"__nneg\"] == 0:\n                out[tgt] = max(out[tgt], gscore * 0.92)\n                out[tgt + \"__conf\"] = max(out[tgt + \"__conf\"], 0.4)\n\n    # Synovitis is frequently visible on the images and absent from the text, so silence\n    # is weak evidence of absence here in a way it is not for other findings. Effusion is\n    # its most reliable textual proxy - the two share a mechanism - so a silent synovitis\n    # inherits a fraction of the effusion evidence instead of falling to the floor.\n    if out[\"Synovitis__npos\"] == 0 and out[\"Synovitis__nneg\"] == 0:\n        out[\"Synovitis\"] = max(out[\"Synovitis\"], 0.28 + 0.45 * (out[\"Effusion\"] - 0.28))\n\n    return out\n# === CELL 8 ===\nfrom __future__ import annotations\n\nimport os\n\nfor _v in (\"OMP_NUM_THREADS\", \"OPENBLAS_NUM_THREADS\", \"MKL_NUM_THREADS\"):\n    os.environ.setdefault(_v, \"4\")\n\nimport gc\nimport hashlib\nimport json\nimport re\nimport time\nimport traceback\nfrom concurrent.futures import ThreadPoolExecutor\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\n# The label extractor is defined in the cells above when this runs as a notebook. As a\n# plain script it is imported from the package source, so the two paths share one\n# definition rather than keeping a copy each.\n\nT0 = time.time()\nSAVE_WEIGHTS = True\nSEED = 2026\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\n\nTARGETS = [\"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\", \"Medial OA\",\n           \"Lateral OA\", \"PF OA\", \"Effusion\", \"Synovitis\", \"Baker's\",\n           \"Contusion\", \"Fracture\"]\n\n\n# The centre crop has to be smaller than the smallest field of view in the corpus or it\n# silently does nothing. Measured over every training series, the acquired field of view\n# (Rows x PixelSpacing) has median 160 mm and runs from 70 to 320: a 160 mm crop is\n# larger than the image in 60% of series and is skipped for all of them, which leaves\n# their physical scale unnormalised. 130 mm is below the field of view of 99.6% of\n# series and still contains the joint.\nCROP_MM = 130.0\n\n# Cache resolution. Everything downstream may downsample from this, so it is set by the\n# most demanding configuration rather than by the default one.\nCACHE_IMG = 336\nGROUP = 3                  # slices per encoder input, stacked as the three channels\nN_GROUP_MAX = 1\nCACHE_FRACTION = 0.45      # share of free memory the pixel cache may take\nCACHE_BUDGET_MAX_GB = 24.0 # hard ceiling regardless of what the machine reports\nCACHE_BUDGET_GB = 12.0     # only the fallback, for a machine with no /proc/meminfo\nTEST_SHARE = 0.30          # floor on the test corpus relative to the training one, since\n                           # the visible test split is a stub and the scored one is not\nHDR_THREADS = 16\nPIX_THREADS = 12\nORDER_THREADS = 32         # slice-ordering is latency-bound on the mount, not CPU-bound\n# Ceiling for the ordering pass. It has to be a ceiling because the pass is hundreds of\n# thousands of small reads over a network mount, so its duration is a property of the\n# mount rather than of the work, and varies between runs that do the same reading. It must not be a tight one: giving up leaves\n# those series in file order, which is uncorrelated with anatomy, and that degradation is\n# silent. So the ceiling sits well above what the pass ordinarily needs: its purpose is\n# to stop the pass consuming the whole run on a slow mount, not to trim the ordinary\n# case, and a ceiling tight enough to bind on a normal day would trade a silent\n# degradation for a saving the run does not need.\nORDER_BUDGET_S = 5400\n\n# Resolution is the axis under test. A feature of width d mm survives resampling only if\n# the pixel pitch is at most d/2, and the pitch here is set by the crop above rather than\n# by the acquired field of view: CROP_MM / P. At 224 px that is 0.58 mm, above the 0.5 mm\n# a 1 mm tear needs; at 336 px it is 0.39 mm and clears it. Both configurations read the\n# same cache, so the comparison isolates the resize.\nRUNS = [\n    {\"name\": \"r224\", \"img\": 224},\n    {\"name\": \"r336\", \"img\": 336},\n]\n\nEPOCHS = 10\nBATCH_STUDIES = 8          # a study is a bag of up to N_SLOT slot images\nAUG_ROT_DEG = 8.0          # rigid jitter; see augment() for why neither flip is used\nAUG_SCALE = 0.08\nAUG_SHIFT = 0.05\nAUG_INTENSITY = 0.10\nLAT_MIN_OFFSET_MM = 20.0   # inside this the side is not readable from geometry; see\n                           # side_from_geometry()\nSLICE_BAND = (0.20, 0.80)  # fraction of the ordered stack read_slot samples across\n\n# --- What a slice IS, as opposed to how many of them there are --------------- #\n#\n# A member is a function of the pixels it was fitted on, and img/crop_mm/slices/band do\n# not determine those pixels by themselves. Four further decisions do, none of them\n# visible in any shape:\n#\n#   order          which slice is the next one along the stack\n#   lat            which knees are mirrored, and on what evidence\n#   slot_fallback  whether a T1 slot may be filled from a series that is not T1\n#   decode_fill    what stands in for a slice that would not decode\n#\n# `native` is the reading derived in the sections below. `legacy` is the reading an\n# imported member was fitted under. A member read under the wrong one loads with every\n# shape matching, runs, and writes a plausible submission computed from the wrong image -\n# so the choice travels with the member and is part of the key that decides which members\n# can share a decode. The legacy rules are reproduced rather than corrected: correcting\n# them would hand that member pixels its weights never saw.\nRULES_NATIVE = {\"order\": \"normal\", \"lat\": \"centre\",\n                \"slot_fallback\": False, \"decode_fill\": \"nearest\"}\nRULES_LEGACY = {\"order\": \"dominant_axis\", \"lat\": \"corner_x\",\n                \"slot_fallback\": True, \"decode_fill\": \"zero\"}\nRULES = dict(RULES_NATIVE)\nLEGACY_LAT_OFFSET_MM = 5.0   # the dead zone the legacy laterality rule was fitted with\n\nLR_HEAD = 1e-3\nLR_BACKBONE = 8e-6         # the encoder is adapted, not retrained\nUNFREEZE_LAST = 6          # trainable transformer blocks, from the output end\nWEIGHT_DECAY = 0.02\nEVAL_BATCH = 8\nTIME_BUDGET = 8.0 * 3600\n\n# Six slots: three planes crossed with the acquisition axes. The fat-suppressed\n# fluid-sensitive series exist for nearly every study; the T1 and the non-suppressed\n# fluid-sensitive series are scarcer, which is what the presence mask is for.\nSLOTS_RECOVERED = [\n    (\"SAG_FLUID_FS\", \"Sagittal\", True, True),\n    (\"COR_FLUID_FS\", \"Coronal\", True, True),\n    (\"AX_FLUID_FS\", \"Axial\", True, True),\n    (\"SAG_FLUID_NOFS\", \"Sagittal\", True, False),\n    (\"COR_T1\", \"Coronal\", False, False),\n    (\"SAG_T1\", \"Sagittal\", False, False),\n]\n\n# The alternative: plane x the single axis the delivered flags carry, ignoring the\n# recovered weighting. Kept as\n# a switch so the choice of slot definition can be varied while everything else is held\n# fixed. Under this scheme a `Struct` slot mixes T1 series with non-fat-suppressed PD/T2\n# series, which carry very different tissue contrast.\nSLOTS_PUBLIC = [\n    (\"SAG_FLUID\", \"Sagittal\", None, True),\n    (\"COR_FLUID\", \"Coronal\", None, True),\n    (\"AX_FLUID\", \"Axial\", None, True),\n    (\"SAG_STRUCT\", \"Sagittal\", None, False),\n    (\"COR_STRUCT\", \"Coronal\", None, False),\n    (\"AX_STRUCT\", \"Axial\", None, False),\n]\n\nSLOT_SCHEME = os.environ.get(\"SLOT_SCHEME\", \"recovered\")\nSLOTS = SLOTS_PUBLIC if SLOT_SCHEME == \"public\" else SLOTS_RECOVERED\nN_SLOT = len(SLOTS)\n\n# How many 384-wide parts the per-slot feature is built from. The encoder emits one\n# vector per token; a slot feature is a fixed summary of that grid, and the summary an\n# imported member was fitted with carries a third part.\nPOOL_PARTS = {\"cls_mean\": 2, \"cls_mean_focal\": 3}\n\n# Which slots an imported member's attention is tilted toward, per diagnosis. Indices are\n# into SLOTS. This is a fixed table rather than a learned parameter, so it is part of that\n# member's definition and has to be reproduced exactly for its weights to mean anything.\nSLOT_PRIOR_TABLE = {\n    \"ACL\": (0, 3, 5), \"MCL\": (1, 4),\n    \"Medial Meniscus\": (0, 1, 3, 4), \"Lateral Meniscus\": (0, 1, 3, 4),\n    \"Medial OA\": (1, 4, 5), \"Lateral OA\": (1, 4, 5),\n    \"PF OA\": (0, 2, 5), \"Effusion\": (0, 2), \"Synovitis\": (0, 2),\n    \"Baker's\": (0,), \"Contusion\": (0, 1, 2), \"Fracture\": (0, 1, 2, 4, 5),\n}\nSLOT_PRIOR_STRENGTH = 0.55\n\nFATSAT_OPTS = {\"FS\", \"FATSAT\", \"FAT_SAT\", \"FSAT\"}\n_SEP = re.compile(r\"[_\\-.]\")\n_FATSAT_RX = re.compile(r\"\\bfs\\b|fatsat|fat sat|\\bstir\\b|\\bspair\\b|\\bspir\\b|\\bwe\\b|\"\n                        r\"water excit|\\btirm\\b|\\bsting\\b|\\bfatsup\\b\")\n_T1_RX = re.compile(r\"\\bt1\\b|\\bt1w\\b\")\n_T2_RX = re.compile(r\"\\bt2\\b|\\bt2w\\b\")\n_PD_RX = re.compile(r\"\\bpd\\b|\\bpdw\\b|proton|\\bdp\\b|dens\")\n\n\n# === CELL 9 ===\ndef log(msg):\n    print(f\"[{time.time() - T0:7.1f}s] {msg}\", flush=True)\n\n\ndef find_root():\n    return Path(\"/kaggle/input/competitions/rsna-knee-abnormality-detection\")\n\n\ndef find_dinov2(variant=\"small\"):\n    return Path(\"/kaggle/input/models/metaresearch/dinov2/pytorch/small/1\")\n\n\nLABEL_COLS = TARGETS + [t + \"__conf\" for t in TARGETS]\n\n\nclass LabelSourceError(RuntimeError):\n    \"\"\"Raised when the labels did not come from where this run intended.\n\n    Every other failure in this file is better survived than reported: a run that dies\n    after the cache is built has spent the expensive half and scores nothing, so the\n    guard around `main` swallows it and leaves the benchmark file behind. This one is\n    the exception. Training on the weaker labels does not look like a failure - it\n    completes, writes a plausible submission, and differs only in a log line - so it has\n    to stop the run rather than be absorbed by a guard designed for crashes.\n    \"\"\"\n\n\ndef find_label_table():\n    \"\"\"Return the hardcoded path to the LLM label table.\"\"\"\n    label_dir = Path(\"/kaggle/input/datasets/pilkwang/rsna-knee-llm-labels\")\n    if label_dir.is_dir():\n        for f in label_dir.iterdir():\n            if f.name.endswith(\".csv\"):\n                return f\n    return None\n\n\ndef label_mount_attached():\n    return Path(\"/kaggle/input/datasets/pilkwang/rsna-knee-llm-labels\").is_dir()\n\n\ndef read_labels(train_df):\n    \"\"\"Labels for every training study, from a mounted table or from the lexicon.\n\n    Studies the mounted table does not cover fall back to the lexicon rather than being\n    dropped, so a partial table degrades coverage instead of losing rows.\n    \"\"\"\n    n = len(train_df)\n    lab = pd.DataFrame([extract(r) for r in train_df[\"Report\"].fillna(\"\")])\n    lab[\"StudyInstanceUID\"] = train_df[\"StudyInstanceUID\"].values\n    lab = lab.set_index(\"StudyInstanceUID\")\n\n    src = find_label_table()\n    if src is None:\n        if label_mount_attached():\n            raise LabelSourceError(\n                \"LABEL SOURCE: a label dataset is mounted but no usable table was found \"\n                \"in it. Falling back to the lexicon here would train on the weaker \"\n                \"labels and say so only in a log line, so the run stops instead.\")\n        log(f\"LABEL SOURCE: lexicon, {n} studies (no table mounted)\")\n        return lab\n\n    tab = pd.read_csv(src).set_index(\"StudyInstanceUID\")\n    missing = [c for c in LABEL_COLS if c not in tab.columns]\n    if missing:\n        raise LabelSourceError(\n            f\"LABEL SOURCE: {src} is missing {len(missing)} expected columns \"\n            f\"(first: {missing[0]!r}). Refusing to fall back silently.\")\n    hit = lab.index.intersection(tab.index)\n    if not len(hit):\n        raise LabelSourceError(\n            f\"LABEL SOURCE: {src} shares no StudyInstanceUID with train.csv.\")\n    log(f\"LABEL SOURCE: {src.name} covers {len(hit)} of {n} studies, \"\n        f\"lexicon for the remaining {n - len(hit)}\")\n    lab.loc[hit, LABEL_COLS] = tab.loc[hit, LABEL_COLS].values\n    return lab\n\n\nROOT = find_root()\nlog(f\"input root: {ROOT}\")\n\n\nIMG = CACHE_IMG            # kept as the name the pixel reader and cache use\n\n\ndef available_gb():\n    \"\"\"Memory this machine will actually lend, read rather than assumed.\n\n    A hardcoded ceiling is a guess about a machine the author is not sitting at, and a\n    guess that is too low costs coverage silently while a guess that is too high ends the\n    run. The machine will say, so it is asked.\n    \"\"\"\n    try:\n        with open(\"/proc/meminfo\") as fh:\n            info = {k.strip(): v for k, v in\n                    (l.split(\":\", 1) for l in fh if \":\" in l)}\n        return int(info[\"MemAvailable\"].split()[0]) / 1024 ** 2\n    except Exception:\n        return CACHE_BUDGET_GB / CACHE_FRACTION      # fall back to the old constant\n\n\ndef plan_cache(n_study, n_test=0):\n    \"\"\"Choose how many slices per slot the memory the machine has will allow.\n\n    The cache is n_study x n_slot x slices x IMG^2 bytes. Coverage is the cheap axis -\n    linear - and resolution the expensive one, so when the budget binds it is the slice\n    count that gives way rather than the pixel grid. Deciding once, from the training\n    corpus size, keeps train and test caches on the same group layout.\n\n    Only a fraction of what is free is taken. The rest is not slack: the encoder, its\n    activations, the pinned batches and the frames all come out of the same pool, and the\n    cache is the one allocation big enough that overshooting it kills the run outright.\n    \"\"\"\n    avail = available_gb()\n    budget = min(avail * CACHE_FRACTION, CACHE_BUDGET_MAX_GB)\n    # Both caches are held at once, and the test half is what the visible run cannot\n    # show: here it is a handful of studies, and at scoring it is the whole hidden set.\n    # Sizing against the training corpus alone therefore passes every run that can be\n    # watched and overruns the one that counts.\n    n_total = n_study + max(n_test, int(TEST_SHARE * n_study))\n    per_slice = n_total * N_SLOT * IMG * IMG\n    afford = int(budget * 1024 ** 3 // max(per_slice, 1))\n    groups = max(1, min(N_GROUP_MAX, afford // GROUP))\n    log(f\"memory: {avail:.1f} GB available, {budget:.1f} GB to the cache; \"\n        f\"sizing for {n_study} train + {n_total - n_study} test studies \"\n        f\"-> {groups} group(s) of {GROUP} = {groups * GROUP} slices per slot\"\n        + (f\" (wanted {N_GROUP_MAX})\" if groups < N_GROUP_MAX else \"\"))\n    return groups\n\n\nN_GROUP = plan_cache(len(pd.read_csv(ROOT / \"train.csv\")),\n                     len(pd.read_csv(ROOT / \"test.csv\")))\nCACHE_SLICES = GROUP * N_GROUP\nlog(f\"cache layout: {N_GROUP} groups x {GROUP} slices = {CACHE_SLICES} per slot\")\n\n\n# === CELL 10 ===\nHDR_TAGS = [\"SeriesDescription\", \"SequenceName\", \"ScanOptions\", \"ScanningSequence\",\n            \"RepetitionTime\", \"EchoTime\", \"Laterality\", \"PixelSpacing\", \"Rows\",\n            \"Columns\", \"RescaleSlope\", \"RescaleIntercept\",\n            # Position and orientation are read from the same header probe() already\n            # opens, so they cost nothing, and they are what recovers the side when the\n            # Laterality tag is absent - which it is for half the studies here.\n            \"ImagePositionPatient\", \"ImageOrientationPatient\"]\n\n\ndef _hdr_vec(s, n):\n    \"\"\"Parse a DICOM multi-value string as stored by probe(): floats joined by `|`.\"\"\"\n    if not isinstance(s, str):\n        return None\n    try:\n        v = [float(x) for x in s.split(\"|\")]\n    except ValueError:\n        return None\n    return np.array(v) if len(v) >= n else None\n\n\ndef side_from_geometry(h):\n    \"\"\"Study -> 'L' / 'R' / None, from where the image sits in the patient.\n\n    `Laterality` (0020,0060) is Type 2C and may legitimately be absent; in this corpus it\n    is missing on exactly half the studies, and the vendors it is missing from are whole\n    vendors rather than scattered series. A study with no tag is not a left knee, but the\n    normalisation upstream treats it as one, so half the corpus was never normalised and\n    the five side-defined targets - the two menisci, the two tibiofemoral compartments\n    and the medial collateral ligament - saw that axis reversed on a large minority of it.\n\n    The patient coordinate system fixes this without the tag: +x is the patient's left, so\n    the centre of a right knee sits at negative x. The centre is used rather than\n    `ImagePositionPatient` itself because that is the corner of the image, which is offset\n    by half a field of view - enough to change the sign on a knee near the midline.\n\n    The median over a study's series is what is thresholded, not a single series: probe()\n    reads one arbitrary slice per series, which on a sagittal stack can sit anywhere\n    across the joint. Studies whose centre falls near the midline are left unresolved\n    rather than guessed - measured against the tagged half, the rule is right 97% of the\n    time overall and no better than chance inside 20 mm.\n    \"\"\"\n    cx = {}\n    for r in h.itertuples(index=False):\n        ipp = _hdr_vec(getattr(r, \"ImagePositionPatient\", None), 3)\n        iop = _hdr_vec(getattr(r, \"ImageOrientationPatient\", None), 6)\n        ps = _hdr_vec(getattr(r, \"PixelSpacing\", None), 2)\n        rows, cols = getattr(r, \"Rows\", None), getattr(r, \"Columns\", None)\n        if ipp is None or iop is None or ps is None or not rows or not cols:\n            continue\n        try:\n            c = ipp[:3] + iop[:3] * ps[1] * float(cols) / 2 + iop[3:6] * ps[0] * float(rows) / 2\n        except (TypeError, ValueError):\n            continue\n        cx.setdefault(r.StudyInstanceUID, []).append(float(c[0]))\n    out = {}\n    for st, xs in cx.items():\n        m = float(np.median(xs))\n        out[st] = None if abs(m) < LAT_MIN_OFFSET_MM else (\"R\" if m < 0 else \"L\")\n    return out\n\n\ndef side_from_corner_x(h):\n    \"\"\"The laterality an imported member was fitted under.\n\n    It thresholds the median raw `ImagePositionPatient` x over a study's series. That is\n    the x of the image *corner*, not of its centre, so it differs from the rule above by\n    up to half a field of view - which is enough to reverse the sign on a knee scanned\n    near the midline. The dead zone is 5 mm rather than 20 mm, so it also commits on\n    studies the rule above leaves unresolved.\n\n    Neither difference changes a shape. Each one decides whether a study is mirrored, and\n    a study mirrored one way at training and the other at inference presents the five\n    side-defined targets with their axis reversed.\n    \"\"\"\n    out = {}\n    for st, g in h.groupby(\"StudyInstanceUID\"):\n        xs = []\n        for r in g.itertuples(index=False):\n            ipp = _hdr_vec(getattr(r, \"ImagePositionPatient\", None), 3)\n            if ipp is not None and np.isfinite(ipp).all():\n                xs.append(float(ipp[0]))\n        if not xs:\n            out[st] = None\n            continue\n        x = float(np.median(xs))\n        # DICOM patient coordinates are LPS: +x is the patient's left.\n        out[st] = None if abs(x) < LEGACY_LAT_OFFSET_MM else (\"R\" if x < 0 else \"L\")\n    return out\n\n\ndef lat_of(h, tag=\"\"):\n    \"\"\"Study -> 'L' / 'R' / None: the tag where it exists, geometry where it does not.\n\n    The tag is present on exactly half the studies here and is sometimes an empty\n    string rather than absent, which is not the same as NaN. Treating the other half\n    as left-sided is what `normalise_laterality` did by omission, so the geometry\n    fallback is not a refinement - it is the difference between normalising half the\n    corpus and normalising all of it.\n    \"\"\"\n    geo = side_from_corner_x(h) if RULES[\"lat\"] == \"corner_x\" else side_from_geometry(h)\n    d, n_tag, n_geo, n_none, n_disagree = {}, 0, 0, 0, 0\n    for st, g in h.groupby(\"StudyInstanceUID\"):\n        v = [str(x).strip().upper() for x in g[\"Laterality\"].dropna()]\n        if RULES[\"lat\"] == \"corner_x\" and \"ImageLaterality\" in g.columns:\n            # The legacy rule reads the second tag too, so a study tagged only there is\n            # resolved from the tag rather than from geometry.\n            v += [str(x).strip().upper() for x in g[\"ImageLaterality\"].dropna()]\n        v = [x[0] for x in v if x and x[0] in (\"L\", \"R\")]\n        side = v[0] if v else None\n        if side is not None:\n            n_tag += 1\n            if geo.get(st) is not None and geo[st] != side:\n                n_disagree += 1\n        else:\n            side = geo.get(st)\n            n_geo += side is not None\n            n_none += side is None\n        d[st] = side\n    log(f\"{tag}laterality: {n_tag} from the tag, {n_geo} from geometry, \"\n        f\"{n_none} unresolved; tag and geometry disagree on {n_disagree} \"\n        f\"({n_disagree / max(n_tag, 1):.1%} of the tagged)\")\n    return d\ndef probe(item):\n    split, study, series, path = item\n    row = {\"split\": split, \"StudyInstanceUID\": study, \"SeriesInstanceUID\": series,\n           \"dir\": path}\n    try:\n        files = sorted(e.name for e in os.scandir(path) if e.name.endswith(\".dcm\"))\n        row[\"files\"] = files\n        row[\"n_slices\"] = len(files)\n        if not files:\n            return row\n        ds = pydicom.dcmread(os.path.join(path, files[len(files) // 2]),\n                             stop_before_pixels=True, force=True)\n        for t in HDR_TAGS:\n            v = getattr(ds, t, None)\n            if v is None:\n                row[t] = None\n            elif isinstance(v, (list, tuple)) or type(v).__name__ == \"MultiValue\":\n                row[t] = \"|\".join(str(x) for x in v)\n            else:\n                row[t] = str(v)\n    except Exception as exc:\n        row[\"err\"] = str(exc)[:120]\n    return row\n\n\ndef walk(split):\n    \"\"\"Every series directory of a split, with one header read per series.\n\n    An absent split returns an empty frame *with the columns annotate expects*. Returning\n    a bare DataFrame looks like the same thing and is not: the next call indexes\n    `SeriesDescription` and raises KeyError, so the branch that exists to survive a\n    missing split is what turns it into a crash.\n    \"\"\"\n    base = ROOT / split\n    items = []\n    if not base.is_dir():\n        return pd.DataFrame(columns=[\"split\", \"StudyInstanceUID\", \"SeriesInstanceUID\",\n                                     \"dir\", \"files\", \"n_slices\"] + HDR_TAGS)\n    for study in os.scandir(base):\n        if study.is_dir():\n            for series in os.scandir(study.path):\n                if series.is_dir():\n                    items.append((split, study.name, series.name, series.path))\n    with ThreadPoolExecutor(max_workers=HDR_THREADS) as pool:\n        rows = list(pool.map(probe, items))\n    return pd.DataFrame(rows)\n\n\ndef annotate(df):\n    \"\"\"Recover fat suppression and pulse-sequence weighting from the header.\"\"\"\n    desc = (df[\"SeriesDescription\"].fillna(\"\") + \" \" + df[\"SequenceName\"].fillna(\"\"))\n    desc = desc.str.lower().str.replace(_SEP, \" \", regex=True)\n\n    opts = df[\"ScanOptions\"].fillna(\"\").str.upper().str.split(\"|\")\n    # GE writes SAT_GEMS for spatial saturation, so ScanOptions must be matched as\n    # exact tokens; a substring test on \"SAT\" fires on non-fat-sat series.\n    opts_fs = opts.apply(lambda ts: any(t.strip() in FATSAT_OPTS for t in ts))\n    df[\"fatsat\"] = desc.str.contains(_FATSAT_RX) | opts_fs\n\n    tr = pd.to_numeric(df[\"RepetitionTime\"], errors=\"coerce\")\n    te = pd.to_numeric(df[\"EchoTime\"], errors=\"coerce\")\n    gre = df[\"ScanningSequence\"].fillna(\"\").str.upper().str.contains(\"GR\")\n    t1, t2, pdw = desc.str.contains(_T1_RX), desc.str.contains(_T2_RX), desc.str.contains(_PD_RX)\n\n    df[\"weight\"] = np.where(t1 & ~t2 & ~pdw, \"T1\",\n                     np.where(t2 & ~pdw, \"T2\",\n                       np.where(pdw, \"PD\",\n                         np.where(gre, \"GRE\",\n                           np.where(tr < 800, \"T1\",\n                             np.where(te > 60, \"T2\",\n                               np.where(tr >= 800, \"PD\", \"UNK\")))))))\n    df[\"fluid\"] = np.isin(df[\"weight\"], [\"PD\", \"T2\"])\n    df[\"px\"] = pd.to_numeric(\n        df[\"PixelSpacing\"].fillna(\"\").str.split(\"|\").str[0].replace(\"\", np.nan),\n        errors=\"coerce\")\n    return df\n\n\n# === CELL 11 ===\ndef pick_slots(series_df, plane_map):\n    \"\"\"One series per slot per study.\n\n    Ties are broken toward the stack with the most slices: a thicker stack samples the\n    joint more densely, and the three-slice sampler below benefits from the margin.\n    \"\"\"\n    series_df = series_df.copy()\n    series_df[\"plane\"] = series_df[\"SeriesInstanceUID\"].map(plane_map)\n    out = {}\n    for study, g in series_df.groupby(\"StudyInstanceUID\"):\n        chosen = {}\n        for name, plane, fluid, fs in SLOTS:\n            sel = (g[\"plane\"] == plane) & (g[\"fatsat\"] == fs)\n            # fluid=None means \"do not condition on weighting\" - the public scheme,\n            # where the single provided flag stands in for both axes at once.\n            if fluid is not None:\n                sel &= (g[\"fluid\"] == fluid)\n            cand = g[sel]\n            # A slot with no series matching its predicate stays empty, and no substitute\n            # is admitted from a neighbouring predicate. Relaxing the weighting to fill a\n            # T1 slot would draw from the pool `SAG_FLUID_NOFS` selects from, since that\n            # pool is what remains once the weighting is dropped: over the training corpus\n            # it would put one series in two slots for 2383 of 4407 studies and leave 56%\n            # of the T1 slot holding PD or T2. The presence mask would then assert a\n            # sequence that was never acquired, and the per-diagnosis softmax of §6 would\n            # divide its attention across two identical slots, giving one acquisition\n            # about twice the weight it carries in a study that holds both. The mask is\n            # there to say a slot is absent, which is what an absent slot is.\n            if len(cand) == 0 and RULES[\"slot_fallback\"] and fluid is False:\n                # The relaxation the paragraph above rejects, reproduced because an\n                # imported member was fitted with its T1 slots filled this way: over half\n                # of that member's training studies had a T1 slot holding a series that\n                # is not T1. Leaving those slots empty would present it with a presence\n                # mask it never saw.\n                cand = g[(g[\"plane\"] == plane) & (~g[\"fatsat\"])]\n            if len(cand):\n                chosen[name] = cand.sort_values(\"n_slices\", ascending=False).iloc[0]\n        out[study] = chosen\n    return out\n\n\n# === CELL 12 ===\nORDER_TAGS = [(0x0020, 0x0032), (0x0020, 0x0037), (0x0020, 0x0013)]\n\n# Series in which at least one sampled slice would not decode. A list rather than a\n# counter because appending is atomic under the reader threads, and reported rather than\n# swallowed: unreported, a decode failure is indistinguishable from a black knee.\nDECODE_FAILED = []\n\n\ndef cache_tag(rules=None):\n    \"\"\"The name a decoded cache is stored under.\n\n    It has to name everything that decides the pixels, not only their dimensions. Two\n    configurations that agree on resolution, slice count, crop and band but disagree on\n    how a slice is chosen produce different arrays of identical shape - so a tag built\n    from the dimensions alone lets the second attach to the first one's file and train\n    against pixels it never asked for, with nothing anywhere reporting a mismatch.\n\n    A native reading keeps the plain name, so caches decoded before the rules existed\n    stay valid; anything else earns a suffix.\n    \"\"\"\n    r = dict(RULES if rules is None else rules)\n    t = (f\"{CACHE_IMG}px_{CACHE_SLICES}sl_{int(CROP_MM)}mm_\"\n         f\"{SLICE_BAND[0]:.2f}-{SLICE_BAND[1]:.2f}\")\n    if {k: r.get(k, v) for k, v in RULES_NATIVE.items()} != RULES_NATIVE:\n        t += \"_\" + hashlib.md5(json.dumps(r, sort_keys=True).encode()).hexdigest()[:6]\n    return t\n\n\ndef _natural_key(name):\n    return tuple(int(x) if x.isdigit() else x.lower()\n                 for x in re.split(r\"(\\d+)\", str(name)))\n\n\ndef _order_dominant_axis(rec):\n    \"\"\"The slice order an imported member was fitted under.\n\n    It sorts on the raw patient coordinate along whichever axis varies most across the\n    stack, rather than on the projection onto the slice normal. The two differ by a sign,\n    not by a formula: measured over this corpus every sagittal series has a slice normal\n    with n_x in [-1.00, -0.98], so p.n is the negative of the raw x this sorts on and the\n    two stacks come out exactly reversed. Because the band sampler truncates rather than\n    rounds, its nine indices are not symmetric about the middle, so nine slices drawn from\n    a twenty-six slice stack under one order share two with the other.\n\n    Missing geometry falls back to `InstanceNumber` and then to a natural sort of the file\n    name, both at the same 80% threshold the imported pipeline used.\n    \"\"\"\n    files, d = rec[\"files\"], rec[\"dir\"]\n    rows = []\n    for pos, f in enumerate(files):\n        ipp = inst = None\n        try:\n            ds = pydicom.dcmread(os.path.join(d, f), force=True, stop_before_pixels=True,\n                                 specific_tags=[\"ImagePositionPatient\", \"InstanceNumber\"])\n            raw = getattr(ds, \"ImagePositionPatient\", None)\n            if raw is not None and len(raw) >= 3:\n                c = np.asarray(raw[:3], dtype=np.float64)\n                if np.isfinite(c).all():\n                    ipp = c\n            n = getattr(ds, \"InstanceNumber\", None)\n            if n is not None:\n                inst = float(n)\n        except Exception:\n            pass\n        rows.append((f, ipp, inst, pos))\n\n    placed = [r for r in rows if r[1] is not None]\n    need = max(2, int(0.8 * len(rows)))\n    if len(placed) >= need:\n        xyz = np.stack([r[1] for r in placed])\n        axis = int(np.argmax(np.ptp(xyz, axis=0)))\n        spare = float(np.nanmedian(xyz[:, axis]))\n        rows.sort(key=lambda r: (float(r[1][axis]) if r[1] is not None else spare,\n                                 r[2] if r[2] is not None else float(\"inf\"), r[3]))\n    elif sum(r[2] is not None for r in rows) >= need:\n        rows.sort(key=lambda r: (r[2] if r[2] is not None else float(\"inf\"), r[3]))\n    else:\n        rows.sort(key=lambda r: _natural_key(r[0]))\n    return [r[0] for r in rows], True\n\n\ndef order_slices(rec):\n    \"\"\"Return the series' files sorted along the through-plane axis.\n\n    A DICOM file name here is a SOP Instance UID, which is assigned arbitrarily. Sorting\n    by it therefore produces an order uncorrelated with anatomy - measured over one\n    series, Spearman between file-name rank and physical position is 0.009, i.e. none.\n    Anything that assumes the file order means something is then operating on noise: the\n    three channels of a \"2.5D\" input are three unrelated views rather than neighbouring\n    slices, \"the middle of the stack\" is a random subset, and reversing slice order to\n    normalise laterality reverses nothing meaningful.\n\n    The physical order is recoverable exactly. Each slice carries its position in patient\n    coordinates and the in-plane axes; projecting the position onto the slice normal\n    gives a signed through-plane coordinate, monotonic along the stack:\n\n        n = r_x  x  r_y ,      k = p . n\n\n    `InstanceNumber` is the fallback. It usually tracks the projection up to sign, but\n    interleaved and multi-echo acquisitions need not number slices in the order they\n    occupy in space - but the projection is signed in patient\n    coordinates, which is what laterality normalisation needs.\n    \"\"\"\n    if RULES[\"order\"] == \"dominant_axis\":\n        return _order_dominant_axis(rec)\n    files, d = rec[\"files\"], rec[\"dir\"]\n    keyed = []\n    for f in files:\n        k = None\n        try:\n            ds = pydicom.dcmread(os.path.join(d, f), force=True, stop_before_pixels=True,\n                                 specific_tags=ORDER_TAGS)\n            iop = np.asarray(ds.ImageOrientationPatient, dtype=float)\n            ipp = np.asarray(ds.ImagePositionPatient, dtype=float)\n            k = float(np.dot(ipp, np.cross(iop[:3], iop[3:])))\n        except Exception:\n            try:\n                k = float(ds.InstanceNumber)\n            except Exception:\n                k = None\n        keyed.append((k, f))\n    if any(k is None for k, _ in keyed):\n        # A series with no usable geometry keeps its arbitrary order; that is worse than\n        # sorting but better than dropping the series, and it is logged as a count.\n        return files, False\n    return [f for _, f in sorted(keyed, key=lambda t: t[0])], True\n\n\ndef read_slot(rec, n_slice=None, out_size=None):\n    \"\"\"`n_slice` physically spread slices from one series, at `out_size` pixels.\n\n    Returns uint8 [n_slice, out, out] normalised per-series to its 1st-99th\n    percentile. Percentiles rather than min/max because MR intensity has no absolute\n    scale and a single bright vessel would otherwise compress the whole dynamic range.\n\n    Reading is the expensive half of this pipeline, so the caller reads once at the\n    largest configuration it needs and derives the smaller ones from the returned buffer\n    rather than re-reading.\n    \"\"\"\n    n_slice = GROUP if n_slice is None else n_slice\n    out_size = IMG if out_size is None else out_size\n    files, d, px = rec.get(\"ordered\") or rec[\"files\"], rec[\"dir\"], rec[\"px\"]\n    n = len(files)\n    if n == 0:\n        return None\n    # Spread the samples over a central band of the stack: the outermost slices of a knee\n    # series are mostly soft tissue outside the joint. The band is a constant rather than\n    # a literal because how much of the stack is worth reading depends on how many slices\n    # are being taken - at three the middle is all that fits, while at sixteen the ends\n    # are worth having, and a Baker cyst sits at the posteromedial end of a sagittal one.\n    lo, hi = int(SLICE_BAND[0] * (n - 1)), int(SLICE_BAND[1] * (n - 1))\n    idx = np.unique(np.linspace(lo, hi, n_slice).astype(int)) if hi > lo else np.array([n // 2])\n    while len(idx) < n_slice:\n        idx = np.append(idx, idx[-1])\n\n    planes = []\n    for i in idx[:n_slice]:\n        try:\n            ds = pydicom.dcmread(os.path.join(d, files[int(i)]), force=True)\n            a = ds.pixel_array.astype(np.float32)\n            sl = float(getattr(ds, \"RescaleSlope\", 1) or 1)\n            ic = float(getattr(ds, \"RescaleIntercept\", 0) or 0)\n            a = a * sl + ic\n        except Exception:\n            a = None                      # no shape is known here; see below\n        planes.append(a)\n\n    # A slice that would not decode has no shape of its own, and inventing one is how a\n    # single unreadable file erases a whole series: a substitute allocated at the resize\n    # target while the decoded slices are still native makes the shape check below take\n    # the substitute as the authority and zero the good slices with it, leaving a black\n    # slot that the presence mask still reports as acquired.\n    #\n    # A failure is instead filled from the nearest slice that did decode - the same\n    # convention the sampler already uses when the band holds fewer distinct slices than\n    # were asked for - and a series where nothing decodes is reported absent, which the\n    # mask can express, rather than black, which it cannot.\n    got = [k for k, p in enumerate(planes) if p is not None]\n    if RULES[\"decode_fill\"] == \"zero\":\n        # What an imported member was fitted with: a failure becomes a zero plane at the\n        # resize target, which the shape check below then propagates to the whole slot.\n        # It is the behaviour the paragraph above describes and rejects, kept here only\n        # because that member's weights were learned against slots blacked out this way.\n        if not got:\n            DECODE_FAILED.append(rec.get(\"SeriesInstanceUID\", d))\n        planes = [np.zeros((out_size, out_size), np.float32) if p is None else p\n                  for p in planes]\n        got = list(range(len(planes)))\n    if not got:\n        DECODE_FAILED.append(rec.get(\"SeriesInstanceUID\", d))\n        return None\n    if len(got) < len(planes):\n        DECODE_FAILED.append(rec.get(\"SeriesInstanceUID\", d))\n        for k, p in enumerate(planes):\n            if p is None:\n                planes[k] = planes[min(got, key=lambda j: abs(j - k))]\n\n    # Slices of one series can still differ in matrix size - multi-echo and some\n    # reformats do - and those are genuinely not stackable.\n    shp = planes[0].shape\n    planes = [p if p.shape == shp else np.zeros(shp, np.float32) for p in planes]\n    vol = np.stack(planes)\n\n    # constant physical extent, then resize: PixelSpacing varies 3.4x across the corpus\n    if px and np.isfinite(px) and px > 0:\n        want = int(round(CROP_MM / px))\n        h, w = shp\n        if 16 < want < min(h, w):\n            cy, cx = h // 2, w // 2\n            half = want // 2\n            vol = vol[:, max(0, cy - half):cy + half, max(0, cx - half):cx + half]\n\n    lo_v, hi_v = np.percentile(vol, [1, 99])\n    vol = np.clip((vol - lo_v) / max(hi_v - lo_v, 1e-6), 0, 1)\n\n    t = torch.from_numpy(np.ascontiguousarray(vol)).unsqueeze(0)\n    t = F.interpolate(t, size=(out_size, out_size), mode=\"bilinear\", align_corners=False)\n    # uint8, not float32. These buffers queue up between the reader threads and the\n    # encoder, and at this size a float32 slot-series is several megabytes. Intensity is\n    # already normalised into [0, 1] here, so eight bits cost nothing that a bilinear\n    # resize has not already cost, and the queue is a quarter the size.\n    return (t.squeeze(0) * 255).round().clamp(0, 255).to(torch.uint8)\n\n\n# === CELL 13 ===\ndef normalise_laterality(img, plane, lat):\n    \"\"\"Map every knee onto a left-knee convention.\n\n    Coronal and axial views mirror under a horizontal flip. Sagittal stacks are not\n    mirror images of each other - the slice order runs medial-to-lateral in opposite\n    directions - so the channel order is reversed instead.\n    \"\"\"\n    if lat != \"R\":\n        return img\n    if plane in (\"Coronal\", \"Axial\"):\n        return torch.flip(img, dims=[-1])\n    return torch.flip(img, dims=[0])\n\n\n# === CELL 14 ===\n# Where the geometric slice order may be remembered between runs. Unset on the platform,\n# because each run gets a fresh machine and there is nothing to remember; set off it,\n# where the same corpus is cached again at every resolution and slice count and the order\n# is a function of neither. It is opt-in so that the scored run's behaviour is decided by\n# the code rather than by whether a file happens to be lying about.\nORDER_CACHE = os.environ.get(\"RSNA_ORDER_CACHE\") or None\n\n\ndef build_cache(slot_map, plane_map, lat_map, tag):\n    \"\"\"Decode every (study, slot) once into an in-memory uint8 array.\n\n    Fine-tuning revisits the same pixels every epoch. Reading them from the mount each\n    time would make the epoch count a function of I/O rather than of learning, so they\n    are decoded once and held as bytes: intensity has already been normalised into\n    [0, 1], and eight bits cost nothing a bilinear resize has not already cost.\n\n    CACHE_SLICES positions are kept per slot, which the training loop reads as N_GROUP\n    groups of GROUP consecutive channels.\n    \"\"\"\n    studies = sorted(slot_map)\n    sidx = {s: i for i, s in enumerate(studies)}\n    cache = np.zeros((len(studies), N_SLOT, CACHE_SLICES, IMG, IMG), np.uint8)\n    mask = np.zeros((len(studies), N_SLOT), np.float32)\n    log(f\"{tag}: cache {cache.shape} = {cache.nbytes / 1024 ** 3:.1f} GB\")\n\n    jobs = [(st, k, plane, slot_map[st][name])\n            for st in studies\n            for k, (name, plane, _, _) in enumerate(SLOTS)\n            if name in slot_map[st]]\n    n_job = len(jobs)\n\n    # Ordering first, and as its own pass. It reads one header per slice of every chosen\n    # series - far more file opens than the decode that follows - and on a network mount\n    # that is latency, not work, so it gets its own wider pool.\n    t_ord = time.time()\n    n_slice_total = sum(len(j[3][\"files\"]) for j in jobs)\n    log(f\"{tag}: ordering {len(jobs)} slot-series ({n_slice_total} slice headers)\")\n    ok = done = 0\n    CHUNK_O = 1024\n\n    # A remembered order, when one is offered. The projection depends on the DICOM\n    # geometry alone, so it is the same at every resolution and every slice count, and\n    # it costs one header read per slice - the largest single cost in this pass. An entry\n    # is validated by the number of files present, so a tree that has changed under it is\n    # recomputed rather than trusted: order is derived data, and a stale entry would be\n    # invisible in the way that matters most.\n    seen = {}\n    if ORDER_CACHE and Path(ORDER_CACHE).is_file():\n        try:\n            import json as _json\n            seen = _json.loads(Path(ORDER_CACHE).read_text())\n        except (OSError, ValueError):\n            seen = {}\n        hit = 0\n        for _, _, _, rec in jobs:\n            e = seen.get(rec[\"SeriesInstanceUID\"])\n            if e and len(e[\"files\"]) == len(rec[\"files\"]):\n                rec[\"ordered\"] = e[\"files\"]\n                ok += int(e[\"good\"])\n                hit += 1\n        jobs = [j for j in jobs if \"ordered\" not in j[3]]\n        log(f\"{tag}: {hit} slot-series ordered from {ORDER_CACHE}, {len(jobs)} to read\")\n\n    with ThreadPoolExecutor(max_workers=ORDER_THREADS) as pool:\n        for c0 in range(0, len(jobs), CHUNK_O):\n            block = jobs[c0:c0 + CHUNK_O]\n            for (_, _, _, rec), (files, good) in zip(\n                    block, pool.map(lambda j: order_slices(j[3]), block)):\n                rec[\"ordered\"] = files\n                ok += int(good)\n                done += 1\n                if ORDER_CACHE:\n                    seen[rec[\"SeriesInstanceUID\"]] = {\"files\": files, \"good\": bool(good)}\n            # The ceiling is whichever comes first: the pass's own budget, or the share\n            # of what is left of the run that it may take. The second is what makes the\n            # first safe to set generously - a mount slow enough to matter cannot spend\n            # the training time, because the budget shrinks as the run does.\n            budget = min(ORDER_BUDGET_S, max(60.0, (TIME_BUDGET - (time.time() - T0)) * 0.35))\n            if time.time() - t_ord > budget:\n                log(f\"{tag}: ordering budget spent at {done}/{len(jobs)}; \"\n                    f\"the rest keep file order\")\n                break\n    if ORDER_CACHE and done:\n        import json as _json\n        _t = Path(ORDER_CACHE).with_suffix(\".tmp\")\n        _t.write_text(_json.dumps(seen))\n        _t.replace(Path(ORDER_CACHE))\n    log(f\"{tag}: ordered {ok}/{n_job} by geometry \"\n        f\"({n_job - ok} kept arbitrary) in {time.time() - t_ord:.0f}s\")\n\n    jobs = [(st, k, plane, slot_map[st][name])\n            for st in studies\n            for k, (name, plane, _, _) in enumerate(SLOTS)\n            if name in slot_map[st]]\n    log(f\"{tag}: decoding {len(jobs)} slot-series\")\n    n_failed_before = len(DECODE_FAILED)\n\n    CHUNK = 512\n    done = 0\n    with ThreadPoolExecutor(max_workers=PIX_THREADS) as pool:\n        for c0 in range(0, len(jobs), CHUNK):\n            block = jobs[c0:c0 + CHUNK]\n            for (st, k, plane, _), img in zip(\n                    block, pool.map(lambda j: read_slot(j[3], CACHE_SLICES, IMG), block)):\n                done += 1\n                if img is None:\n                    continue\n                cache[sidx[st], k] = normalise_laterality(img, plane,\n                                                          lat_map.get(st)).numpy()\n                mask[sidx[st], k] = 1.0\n            if done % 4096 < CHUNK:\n                log(f\"  {tag} {done}/{len(jobs)}\")\n            if time.time() - T0 > TIME_BUDGET:\n                log(f\"  {tag}: time budget reached during decode\")\n                break\n    n_failed = len(DECODE_FAILED) - n_failed_before\n    log(f\"{tag}: {int(mask.sum())}/{len(jobs)} slots filled\"\n        + (f\"; {n_failed} series had a slice that would not decode\" if n_failed else \"\"))\n    gc.collect()\n    return studies, cache, mask\n# === CELL 15 ===\nclass SlotHead(nn.Module):\n    \"\"\"Per-diagnosis attention over the slot embeddings of one study.\n\n    Each finding is read on particular sequences - cruciates sagittally, collateral\n    ligaments and the meniscal body coronally, patellar cartilage axially - so pooling\n    the slots identically would dilute the one that carries the evidence with the rest.\n\n    The aggregation is deliberately this simple. With a study-level label there is no\n    signal telling the model which part of a study matters, so extra attention\n    parameters below the slot level would have nothing to learn from and would spend\n    their capacity fitting noise.\n    \"\"\"\n\n    def __init__(self, dim, n_slot, n_out, hidden=256, p=0.2, prior=False):\n        super().__init__()\n        self.proj = nn.Sequential(nn.LayerNorm(dim), nn.Linear(dim, hidden), nn.GELU())\n        self.slot_emb = nn.Parameter(torch.randn(n_slot, hidden) * 0.02)\n        self.query = nn.Parameter(torch.randn(n_out, hidden) * 0.02)\n        self.drop = nn.Dropout(p)\n        self.out = nn.Linear(hidden, n_out)\n        self.hidden = hidden\n        p_ = torch.zeros(n_out, n_slot)\n        if prior and n_slot == len(SLOTS) and n_out == len(TARGETS):\n            for t, slots in SLOT_PRIOR_TABLE.items():\n                if t in TARGETS:\n                    p_[TARGETS.index(t), list(slots)] = SLOT_PRIOR_STRENGTH\n        self.prior = prior\n        if prior:\n            self.register_buffer(\"slot_prior\", p_)\n\n    def forward(self, x, mask):\n        h = self.proj(x) + self.slot_emb\n        att = torch.einsum(\"bsh,oh->bos\", h, self.query) / self.hidden ** 0.5\n        if self.prior:\n            att = att + self.slot_prior.unsqueeze(0)\n        att = att.masked_fill(mask.unsqueeze(1) < 0.5, -1e4).softmax(-1)\n        ctx = self.drop(torch.einsum(\"bos,bsh->boh\", att, h))\n        return (ctx * self.out.weight.unsqueeze(0)).sum(-1) + self.out.bias\n\n\nclass Model(nn.Module):\n    \"\"\"Encoder plus head, trained end to end.\n\n    A study arrives as a bag of slot images. The bag is flattened for the encoder and\n    folded back before the head, so the encoder never sees the study structure and the\n    head never sees pixels.\n    \"\"\"\n\n    def __init__(self, backbone, dim, pool=\"cls_mean\", prior=False):\n        super().__init__()\n        self.backbone = backbone\n        self.pool = pool\n        self.head = SlotHead(dim * POOL_PARTS[pool], N_SLOT, len(TARGETS), prior=prior)\n        self.register_buffer(\"mean\", torch.tensor([0.485, 0.456, 0.406]).view(1, 3, 1, 1))\n        self.register_buffer(\"std\", torch.tensor([0.229, 0.224, 0.225]).view(1, 3, 1, 1))\n\n    def forward(self, imgs, mask, img_size=None):\n        B, S = imgs.shape[:2]\n        x = imgs.reshape(B * S, *imgs.shape[2:]).float().div_(255.0)\n        if img_size is not None and img_size != x.shape[-1]:\n            x = F.interpolate(x, size=(img_size, img_size), mode=\"bilinear\",\n                              align_corners=False)\n        x = (x - self.mean) / self.std\n        out = self.backbone(pixel_values=x).last_hidden_state\n        patch = out[:, 1:]\n        parts = [out[:, 0], patch.mean(1)]\n        if self.pool == \"cls_mean_focal\":\n            k = max(1, patch.shape[1] // 8)\n            parts.append(patch.topk(k, dim=1).values.mean(1))\n        feat = torch.cat(parts, dim=1).reshape(B, S, -1)\n        return self.head(feat, mask)\n\n\ndef build_model(unfreeze_last, source=None, variant=\"small\", pool=\"cls_mean\",\n                prior=False):\n    \"\"\"Load the encoder and open the last `unfreeze_last` blocks for training.\"\"\"\n    from transformers import AutoModel\n    p = source if source is not None else find_dinov2(variant)\n    if p is None:\n        raise FileNotFoundError(\"DINOv2 weights not attached\")\n    bb = AutoModel.from_pretrained(str(p))\n    n_layer = len(bb.encoder.layer)\n    for prm in bb.parameters():\n        prm.requires_grad = False\n    for blk in bb.encoder.layer[max(0, n_layer - unfreeze_last):]:\n        for prm in blk.parameters():\n            prm.requires_grad = True\n    for prm in bb.layernorm.parameters():\n        prm.requires_grad = True\n    dim = bb.config.hidden_size\n    trainable = sum(p.numel() for p in bb.parameters() if p.requires_grad)\n    log(f\"backbone: {n_layer} blocks, last {unfreeze_last} trainable \"\n        f\"({trainable / 1e6:.1f}M params), feature dim {dim * POOL_PARTS[pool]}\")\n    return Model(bb, dim, pool=pool, prior=prior)\n\n\nclass WeightsError(RuntimeError):\n    pass\n\n\ndef check_fingerprint(model, dev, img_size, fp, tag=\"\"):\n    \"\"\"Verify that a loaded model produces the output its author saw.\n\n    PyTorch has no ABI. A model definition is arbitrary code: any change to it changes\n    what the state dict is loaded into, and typically loads without error, runs without\n    error and returns garbage. There is no other way to tell whether the Model and\n    SlotHead classes in this file are what the loaded weights were fitted for.\n\n    The fingerprint is `sha256(pred.numpy().tobytes())` of a fixed tensor. It is stored\n    in the checkpoint alongside the state dict.\n    \"\"\"\n    model.eval()\n    with torch.no_grad():\n        x = torch.linspace(-2.1, 2.1, 1 * N_SLOT * 3 * img_size * img_size)\n        x = x.view(1, N_SLOT, 3, img_size, img_size).to(dev)\n        mask = torch.ones(1, N_SLOT, dtype=torch.float32, device=dev)\n        pred = model(x, mask, img_size)\n        h = hashlib.sha256(np.ascontiguousarray(pred.cpu().numpy()).tobytes()).hexdigest()\n    if h != fp:\n        raise WeightsError(f\"{tag}fingerprint mismatch: got {h[:8]}, expected {fp[:8]}. \"\n                           f\"The architecture in this script is not what the weights \"\n                           f\"were fitted against. They will return garbage.\")\n\n\n# === CELL 16 ===\ndef augment(vol, mask):\n    # Neither flip is used. 1: \"Medial\" and \"Lateral\" are targets; normalising laterality\n    # means the same side of the image is medial for every knee in the batch, and flipping\n    # it horizontally undoes that. 2: The slice order is sorted to physical space;\n    # flipping that reverses the joint, and the targets are not symmetric in that axis.\n    b, s, c, h, w = vol.shape\n    vol = vol.float() / 255.0\n\n    rot = (torch.rand(b, s) * 2 - 1) * AUG_ROT_DEG\n    sc = 1.0 + (torch.rand(b, s) * 2 - 1) * AUG_SCALE\n    dx = (torch.rand(b, s) * 2 - 1) * AUG_SHIFT\n    dy = (torch.rand(b, s) * 2 - 1) * AUG_SHIFT\n\n    for i in range(b):\n        for j in range(s):\n            if mask[i, j] == 0:\n                continue\n            theta = torch.zeros(2, 3)\n            a = rot[i, j] * 3.14159 / 180.0\n            theta[0, 0] = torch.cos(a) / sc[i, j]\n            theta[0, 1] = -torch.sin(a) / sc[i, j]\n            theta[0, 2] = dx[i, j]\n            theta[1, 0] = torch.sin(a) / sc[i, j]\n            theta[1, 1] = torch.cos(a) / sc[i, j]\n            theta[1, 2] = dy[i, j]\n\n            grid = F.affine_grid(theta.unsqueeze(0), [1, c, h, w], align_corners=False)\n            vol[i, j] = F.grid_sample(vol[i, j].unsqueeze(0), grid, align_corners=False,\n                                      padding_mode=\"reflection\").squeeze(0)\n\n    # independent intensity shifts per slot\n    shift = (torch.rand(b, s, 1, 1, 1) * 2 - 1) * AUG_INTENSITY\n    vol = torch.clamp(vol + shift, 0.0, 1.0)\n\n    vol[:, :, 0] = (vol[:, :, 0] - 0.485) / 0.229\n    vol[:, :, 1] = (vol[:, :, 1] - 0.456) / 0.224\n    vol[:, :, 2] = (vol[:, :, 2] - 0.406) / 0.225\n    return vol\n\n\ndef take_group(cache, mask, idx, g, dev, do_aug):\n    # cache: [study, slot, slice, H, W]\n    b = len(idx)\n    s0 = g * GROUP\n    vol = torch.from_numpy(cache[idx, :, s0:s0 + GROUP])\n    m = torch.from_numpy(mask[idx]).to(dev)\n\n    if do_aug:\n        vol = augment(vol, m)\n    else:\n        vol = vol.float() / 255.0\n        vol[:, :, 0] = (vol[:, :, 0] - 0.485) / 0.229\n        vol[:, :, 1] = (vol[:, :, 1] - 0.456) / 0.224\n        vol[:, :, 2] = (vol[:, :, 2] - 0.406) / 0.225\n    return vol.to(dev), m\n\n\ndef bce_loss(pred, target, conf):\n    return F.binary_cross_entropy_with_logits(pred, target, weight=conf)\n\n\n# === CELL 17 ===\ndef macro_auc(pred_df, lab_df, label_cols):\n    \"\"\"The competition metric, computed exactly as the host computes it.\n\n    A rank-mean ensemble needs probabilities, not predictions. The scorer is written to\n    the spec of a 2017 paper (\"ALROC\") where ties break in favour of the negative class,\n    and the host implementation translates that directly into sorting on (p, 1-y) and\n    integrating.\n\n    A study scored with p=0.0 is therefore ranked below a study scored with p=0.0 and\n    y=1.0. A prediction of all zeros will sort the few positives to the top, so it is a\n    perfect submission; one of all ones sorts them to the bottom, so it scores zero. A\n    network that learns to emit only its bias terms receives a score that depends heavily\n    on what they are - and the metric will report that a member trained against an early\n    stop has \"lost its skill\" when its only fault is returning a constant that the\n    tiebreaker penalises.\n\n    Adding vanishing uniform noise before scoring breaks the ties randomly and returns the\n    metric to what it claims to be, so that what happened in training can be seen.\n    \"\"\"\n    aucs = {}\n    for c in label_cols:\n        p = pred_df[c].values.copy()\n        y = lab_df[c].values\n        mask = (lab_df[c + \"__conf\"] > 0.5).values\n        if mask.sum() < 2:\n            aucs[c] = np.nan\n            continue\n        p = p[mask]\n        y = y[mask]\n        # the tie-breaking fix\n        p += np.random.uniform(0, 1e-12, size=len(p))\n\n        idx = np.argsort(p)\n        y_sorted = y[idx]\n        tpr = np.cumsum(y_sorted) / max(y_sorted.sum(), 1e-6)\n        fpr = np.cumsum(1 - y_sorted) / max((1 - y_sorted).sum(), 1e-6)\n        # integrate the curve\n        aucs[c] = np.sum(np.diff(fpr) * tpr[:-1])\n\n    v = [x for x in aucs.values() if np.isfinite(x)]\n    return np.mean(v) if v else np.nan, aucs\n\n\n# === CELL 18 ===\nTTA_OVERLAP = 1\n\n\ndef window_starts(n_slice, group_size):\n    \"\"\"Indices of the slices that start a TTA window.\"\"\"\n    starts = []\n    i = 0\n    while i + group_size <= n_slice:\n        starts.append(i)\n        i += group_size - TTA_OVERLAP\n    if not starts or starts[-1] + group_size < n_slice:\n        starts.append(max(0, n_slice - group_size))\n    return list(sorted(set(starts)))\n\n\ndef predict_member(model, cache, mask, idx, dev, img_size, starts=None):\n    \"\"\"Inference for one model, one pass.\n\n    The cache was built with N_GROUP groups of GROUP slices. A training batch reads one\n    group. Inference reads every group, and averages the predictions in logit space: it\n    is test-time augmentation (TTA) along the through-plane axis, rolling the 3-channel\n    window across the ordered stack.\n\n    If `starts` is provided, those are the window boundaries. Without it, the member\n    takes the non-overlapping N_GROUP windows it was trained with.\n    \"\"\"\n    model.eval()\n    all_pred = []\n    n_study, n_slot, n_sl, _, _ = cache.shape\n    starts = starts or [g * GROUP for g in range(N_GROUP)]\n\n    with torch.no_grad():\n        for i in range(0, len(idx), EVAL_BATCH):\n            batch_idx = idx[i:i + EVAL_BATCH]\n            m = torch.from_numpy(mask[batch_idx]).to(dev)\n\n            acc = None\n            for s0 in starts:\n                vol = torch.from_numpy(np.ascontiguousarray(\n                    cache[batch_idx, :, s0:s0 + GROUP])).to(dev)\n                with torch.autocast(\"cuda\", enabled=dev.type == \"cuda\"):\n                    z = model(vol, m, img_size).float()\n                acc = z if acc is None else acc + z\n\n            all_pred.append(torch.sigmoid(acc / len(starts)).cpu().numpy())\n\n    return np.concatenate(all_pred) if all_pred else np.zeros((0, len(TARGETS)), np.float32)\n\n\ndef write_submission(pred, studies, test_df, path):\n    \"\"\"Write one submission file from a prediction matrix.\n\n    Predictions are converted to per-column ranks first: the metric reads only order, so\n    ranks discard nothing, and they make files from different configurations directly\n    comparable and safe to average.\n    \"\"\"\n    sub = pd.DataFrame(pd.DataFrame(pred).rank(pct=True).values, columns=TARGETS)\n    sub.insert(0, \"StudyInstanceUID\", studies)\n    sub = test_df[[\"StudyInstanceUID\"]].merge(sub, on=\"StudyInstanceUID\", how=\"left\")\n    sub[TARGETS] = sub[TARGETS].fillna(0.5)\n    sub.to_csv(path, index=False)\n    return sub\n\n\ndef write_benchmark_submission():\n    \"\"\"Write the 0.5 benchmark file immediately.\"\"\"\n    t = pd.read_csv(ROOT / \"test.csv\")\n    for c in TARGETS:\n        t[c] = 0.5\n    t.to_csv(\"submission.csv\", index=False)\n\n\ndef adopt_config_globals(cfg):\n    \"\"\"Overwrite the module constants with the ones this imported member needs.\n\n    Those constants decide the pixel shapes and the model architecture, so a mismatch\n    builds a cache the model cannot read or a model the state dict cannot load into. The\n    fingerprint catches the latter, but the former fails silently.\n    \"\"\"\n    global IMG, CROP_MM, GROUP, N_GROUP, CACHE_SLICES, SLICE_BAND\n    global SLOTS, SLOT_SCHEME, N_SLOT, RULES\n\n    IMG = cfg[\"img\"]\n    CROP_MM = cfg[\"crop_mm\"]\n    GROUP = cfg[\"group\"]\n    N_GROUP = cfg.get(\"n_group\", cfg.get(\"groups\", 1))\n    CACHE_SLICES = cfg[\"slices\"]\n    SLICE_BAND = tuple(cfg[\"band\"])\n    if \"rules\" in cfg:\n        RULES = cfg[\"rules\"]\n    if \"slots\" in cfg:\n        SLOT_SCHEME = cfg[\"slots\"]\n        SLOTS = SLOTS_PUBLIC if SLOT_SCHEME == \"public\" else SLOTS_RECOVERED\n        N_SLOT = len(SLOTS)\n\n\ndef main():\n    write_benchmark_submission()\n    \n    dev = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n    log(f'Device: {dev}')\n    if dev.type == 'cuda':\n        log(f'GPU: {torch.cuda.get_device_name()}')\n    \n    test_df = pd.read_csv(ROOT / 'test.csv')\n    test_series = pd.read_csv(ROOT / 'test_series.csv')\n    plane_map = dict(zip(test_series['SeriesInstanceUID'],\n                         test_series['Anatomical_Plane']))\n    \n    log('header pass: test')\n    hte = annotate(walk('test_series'))\n    log(f'  {len(hte)} test series')\n    \n    # All weight package locations\n    WEIGHT_DIRS = [\n        Path('/kaggle/input/datasets/pilkwang/rsna-knee-weights'),\n        Path('/kaggle/input/notebooks/kush3690/training-new-method/weights'),\n    ]\n    \n    # Collect all members from all packages\n    all_members = []\n    for wdir in WEIGHT_DIRS:\n        manifest_path = wdir / 'manifest.json'\n        if not manifest_path.is_file():\n            # Try looking in subdirectories\n            for p in wdir.rglob('manifest.json'):\n                manifest_path = p\n                wdir = p.parent\n                break\n        if not manifest_path.is_file():\n            log(f'WARNING: No manifest.json found in {wdir}, skipping')\n            continue\n        man = json.loads(manifest_path.read_text())\n        members = man['members']\n        log(f'Package {wdir}: {len(members)} member(s)')\n        for m in members:\n            m['_base_dir'] = str(wdir)\n        all_members.extend(members)\n    \n    log(f'Total members to ensemble: {len(all_members)}')\n    \n    if not all_members:\n        log('ERROR: No weights found!')\n        return\n    \n    # Group members by pixel_group\n    groups = {}\n    for m in all_members:\n        groups.setdefault(m['pixel_group'], []).append(m)\n    \n    per_member = []\n    \n    for gi, (key, gm) in enumerate(groups.items(), 1):\n        cfg = json.loads(key)\n        adopt_config_globals(cfg)\n        log(f'Decode group {gi}/{len(groups)}: {cfg[\"img\"]}px x {cfg[\"slices\"]} slices, '\n            f'crop {cfg[\"crop_mm\"]} mm -> {len(gm)} member(s)')\n        \n        # Build test cache for this pixel group\n        slots_te = pick_slots(hte, plane_map)\n        st_te, Cte, Mte = build_cache(slots_te, plane_map,\n                                       lat_of(hte, 'test '), f'test g{gi}')\n        idx = np.arange(len(st_te))\n        starts = window_starts(Cte.shape[2], GROUP)\n        \n        for m in gm:\n            base_dir = Path(m['_base_dir'])\n            ck_path = base_dir / m['file']\n            if not ck_path.is_file():\n                log(f'  WARNING: {ck_path} not found, skipping')\n                continue\n            \n            t0 = time.time()\n            ck = torch.load(ck_path, map_location='cpu', weights_only=False)\n            \n            model = build_model(\n                int(m['config']['unfreeze_last']),\n                variant=m['config']['variant'],\n                pool=m.get('config', {}).get('pool', 'cls_mean'),\n                prior=bool(m.get('config', {}).get('prior', False))\n            ).to(dev)\n            model.load_state_dict(ck['model'])\n            \n            # Check fingerprint if available\n            if 'fingerprint' in ck:\n                try:\n                    check_fingerprint(model, dev, IMG, ck['fingerprint'], tag=f\"{m['id']}: \")\n                except WeightsError as e:\n                    log(f'  WARNING: {e}')\n            \n            p = predict_member(model, Cte, Mte, idx, dev, IMG, starts=starts)\n            per_member.append({\n                'id': m['id'],\n                'ids': st_te,\n                'pred': p,\n                'holdout': m.get('holdout')\n            })\n            \n            log(f'  {m[\"id\"]} fold {m.get(\"fold\", \"?\")}:  predicted {len(idx)} studies '\n                f'over {len(starts)} window(s) in {time.time()-t0:.0f}s')\n            \n            del model, ck\n            gc.collect()\n            if dev.type == 'cuda':\n                torch.cuda.empty_cache()\n        \n        del Cte, Mte\n        gc.collect()\n    \n    # Rank-mean ensemble across ALL members\n    all_ids = sorted({s for m in per_member for s in m['ids']})\n    pos = {s: i for i, s in enumerate(all_ids)}\n    acc = np.zeros((len(all_ids), len(TARGETS)), np.float64)\n    for m in per_member:\n        r = pd.DataFrame(m['pred']).rank(pct=True).to_numpy()\n        acc[[pos[s] for s in m['ids']]] += r\n    acc /= max(len(per_member), 1)\n    \n    sub = write_submission(acc, all_ids, test_df, 'submission.csv')\n    log(f'submission.csv = rank mean of {len(per_member)} member(s); {sub.shape}; '\n        f'nulls {int(sub[TARGETS].isna().sum().sum())}')\n    \n    # Print summary\n    log('---- Ensemble Summary ----')\n    for m in per_member:\n        log(f\"  {m['id']:45s} holdout {m.get('holdout', 'N/A')}\")\n    log(f'Total members ensembled: {len(per_member)}')\n    log('done')\n\n\nif __name__ == \"__main__\":\n    try:\n        main()\n    except Exception:\n        traceback.print_exc()\n        write_benchmark_submission()\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-08-10T07:16:34.058726Z","iopub.execute_input":"2026-08-10T07:16:34.059034Z","iopub.status.idle":"2026-08-10T07:17:11.700335Z","shell.execute_reply.started":"2026-08-10T07:16:34.05901Z","shell.execute_reply":"2026-08-10T07:17:11.699771Z"}},"outputs":[],"execution_count":null}]}