{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.13"},"papermill":{"default_parameters":{},"duration":5280.157318,"end_time":"2026-08-07T09:09:44.39431+00:00","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-08-07T07:41:44.236992+00:00","version":"2.7.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"111d184820904602b0906a283f18ccd4":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_500bb8fdaa9349e1b7c4aa08f56f990a","max":223,"min":0,"orientation":"horizontal","style":"IPY_MODEL_1a301c998ac9489888ce71ba932e0a7d","tabbable":null,"tooltip":null,"value":223}},"14a63ad675004d549a1e64de293aa2f1":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"1a301c998ac9489888ce71ba932e0a7d":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"1bb9bf6e6a8c4a32b2b87fb97ffd2941":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_1ecd3843df1344b3b72f4b344554ef95","max":223,"min":0,"orientation":"horizontal","style":"IPY_MODEL_5c537283c6ed443481fa4556300db964","tabbable":null,"tooltip":null,"value":223}},"1ecd3843df1344b3b72f4b344554ef95":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"43116046a088428988d014d0c6087e35":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"46c766c0c64544458c513e35248765fd":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"482e8d74e5e44a7c8d6c26abc5a56d25":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"4c4ed6fd6f3141a2970bd3bc1f7efa83":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"500bb8fdaa9349e1b7c4aa08f56f990a":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"52d88633b8414c5191ed434582aad4d4":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_61ab85ff5dcd49469e51a7e627d3f73a","placeholder":"​","style":"IPY_MODEL_7aa52c5f84a1447d9b8eaa7ab059c8a5","tabbable":null,"tooltip":null,"value":" 223/223 [00:00&lt;00:00, 917.80it/s, Materializing param=layernorm.weight]"}},"56ed0b3c74304953a027b132d979de35":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"5c537283c6ed443481fa4556300db964":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"61ab85ff5dcd49469e51a7e627d3f73a":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"73555649893c4f9cbaa9956ae04de6af":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"7952be1923aa41beb7e99e6fd924a6b6":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_14a63ad675004d549a1e64de293aa2f1","placeholder":"​","style":"IPY_MODEL_4c4ed6fd6f3141a2970bd3bc1f7efa83","tabbable":null,"tooltip":null,"value":"Loading weights: 100%"}},"7aa52c5f84a1447d9b8eaa7ab059c8a5":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"9371191494404498a53f7f0e6f6c9f1c":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_56ed0b3c74304953a027b132d979de35","placeholder":"​","style":"IPY_MODEL_482e8d74e5e44a7c8d6c26abc5a56d25","tabbable":null,"tooltip":null,"value":" 223/223 [00:00&lt;00:00, 1116.26it/s, Materializing param=layernorm.weight]"}},"b4ef7cd738fa4940b8ed75c945e51022":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_46c766c0c64544458c513e35248765fd","placeholder":"​","style":"IPY_MODEL_43116046a088428988d014d0c6087e35","tabbable":null,"tooltip":null,"value":"Loading weights: 100%"}},"b542e7e3f8c54e4faff6a8f014629261":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_7952be1923aa41beb7e99e6fd924a6b6","IPY_MODEL_1bb9bf6e6a8c4a32b2b87fb97ffd2941","IPY_MODEL_9371191494404498a53f7f0e6f6c9f1c"],"layout":"IPY_MODEL_cc713193efa94171b5e37aed76bc1902","tabbable":null,"tooltip":null}},"cc713193efa94171b5e37aed76bc1902":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"d9733e30cfef4f5d800a854f25457864":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_b4ef7cd738fa4940b8ed75c945e51022","IPY_MODEL_111d184820904602b0906a283f18ccd4","IPY_MODEL_52d88633b8414c5191ed434582aad4d4"],"layout":"IPY_MODEL_73555649893c4f9cbaa9956ae04de6af","tabbable":null,"tooltip":null}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"cover-00","cell_type":"code","source":"from pathlib import Path\nfrom IPython.display import Image, display\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2026-08-07T16:36:33.059929Z","iopub.execute_input":"2026-08-07T16:36:33.060209Z","iopub.status.idle":"2026-08-07T16:36:33.069274Z","shell.execute_reply.started":"2026-08-07T16:36:33.06018Z","shell.execute_reply":"2026-08-07T16:36:33.06822Z"},"papermill":{"duration":435.790366,"end_time":"2026-08-07T07:49:02.702706+00:00","exception":false,"start_time":"2026-08-07T07:41:46.91234+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"code-06","cell_type":"code","source":"from __future__ import annotations\n\nimport re\nimport unicodedata\n\nTARGETS = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\",\n    \"Medial OA\", \"Lateral OA\", \"PF OA\", \"Effusion\",\n    \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\",\n]\n\n_PRE = str.maketrans({\n    \"ı\": \"i\", \"İ\": \"i\", \"I\": \"i\", \"ß\": \"ss\", \"đ\": \"d\", \"Đ\": \"d\",\n    \"ø\": \"o\", \"Ø\": \"o\", \"æ\": \"ae\", \"Æ\": \"ae\",\n})\n\n\ndef normalize(text: str) -> str:\n    \"\"\"Fold case, diacritics and separators; keep Greek and Cyrillic letters.\n\n    NFKD decomposition strips Latin accents and Greek tonos alike (ά -> α), which is what\n    we want: reports are inconsistent about accents. It also maps the MICRO SIGN U+00B5\n    to a real mu, which matters because most Greek reports here use the wrong codepoint.\n    \"\"\"\n    if not isinstance(text, str):\n        return \"\"\n    text = text.translate(_PRE).lower()\n    text = unicodedata.normalize(\"NFKD\", text)\n    text = \"\".join(ch for ch in text if not unicodedata.combining(ch))\n    text = text.replace(\"­\", \"\")                    # soft hyphen\n    text = re.sub(r\"[_\\-/\\\\]+\", \" \", text)\n    text = re.sub(r\"[ \\t]+\", \" \", text)\n    return text\n\n\n_SENT_SPLIT = re.compile(r\"(?<=[.;!?])\\s+|\\n+\")\n\n\ndef clauses(text: str):\n    \"\"\"Split into clauses, then attach `header:` lines to the value that follows.\n\n    A report line reading `Fractures :` followed by `Aucune.` is one statement. Splitting\n    on punctuation alone separates the anatomy from its negation and flips the label.\n    \"\"\"\n    norm = normalize(text)\n    raw = [c.strip() for c in _SENT_SPLIT.split(norm) if c and c.strip()]\n\n    merged = []\n    for i, c in enumerate(raw):\n\n        if c.endswith(\":\") and len(c.split()) <= 14 and i + 1 < len(raw):\n            merged.append(c + \" \" + raw[i + 1])\n        else:\n            merged.append(c)\n    out = []\n    for c in merged:\n        out.append(c)\n        if len(c.split()) > 25:\n            out.extend(p.strip() for p in c.split(\",\") if len(p.split()) > 2)\n    return out\n\n\ndef _rx(*alts: str) -> re.Pattern:\n    return re.compile(\"|\".join(alts))\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:03.07109Z","iopub.status.busy":"2026-08-07T07:49:03.070134Z","iopub.status.idle":"2026-08-07T07:49:03.080879Z","shell.execute_reply":"2026-08-07T07:49:03.080193Z"},"papermill":{"duration":0.045221,"end_time":"2026-08-07T07:49:03.082377+00:00","exception":false,"start_time":"2026-08-07T07:49:03.037156+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-07","cell_type":"code","source":"NEGATION = _rx(\n    r\"\\bno\\b\", r\"\\bnot\\b\", r\"\\bwithout\\b\", r\"\\bnegative for\\b\", r\"\\babsence\\b\",\n    r\"\\bno evidence\\b\", r\"\\bunremarkable\\b\", r\"\\bfree of\\b\", r\"\\bnone\\b\", r\"\\bnil\\b\",\n    r\"\\bsin\\b\", r\"\\bno hay\\b\", r\"\\bausencia\\b\", r\"\\bausentes?\\b\",\n    r\"\\bpas de\\b\", r\"\\bsans\\b\", r\"\\baucune?\\b\", r\"\\babsence\\b\",\n    r\"\\bgeen\\b\", r\"\\bzonder\\b\", r\"\\bniet\\b\",\n    r\"\\bkeine?\\b\", r\"\\bohne\\b\", r\"\\bnicht\\b\",\n    r\"\\byok\\b\", r\"\\byoktur\\b\", r\"izlenmemekte\", r\"saptanmadi\", r\"\\bdegil\\b\",\n    r\"gozlenmemekte\", r\"mevcut degil\", r\"eslik etmiyor\", r\"\\bizlenmedi\\b\",\n    r\"\\bnema\\b\", r\"\\bbez\\b\", r\"\\bnisu\\b\", r\"\\bnije\\b\",\n    r\"\\bδεν\\b\", r\"\\bχωρις\\b\", r\"ουδεν\",\n    r\"\\bбез\\b\", r\"\\bне\\b\", r\"липсва\", r\"\\bняма\\b\",\n)\n\nNORMALITY = _rx(\n    r\"\\bnormal\", r\"\\bintact\\b\", r\"\\bpreserved\\b\", r\"\\bwithin normal limits\\b\",\n    r\"limites normales\", r\"\\bconservad\", r\"\\bintegr\", r\"\\bnormales\\b\",\n    r\"\\bdoga(l|ll)\\b\", r\"korunmus\", r\"\\bnormaldir\\b\", r\"olagan\",\n    r\"\\buredn\", r\"\\bocuvan\", r\"\\bodrzan\", r\"\\bintakt\",\n    r\"φυσιολογικ\", r\"ακεραι\",\n    r\"unauffallig\", r\"regelrecht\", r\"\\bintakt\\b\",\n    r\"нормал\", r\"запазен\", r\"съхранен\", r\"\\bбез особености\\b\",\n    r\"\\bgaaf\\b\", r\"\\bnormaal\\b\",\n)\n\nUNCERTAIN = _rx(\n    r\"\\bpossible\\b\", r\"\\bprobable\\b\", r\"\\bsuspicious\\b\", r\"\\bsuspected\\b\",\n    r\"cannot (be )?exclude\", r\"\\bmay\\b\", r\"\\bquestionable\\b\", r\"\\bequivocal\\b\",\n    r\"\\bposible\\b\", r\"sin criterios categoricos\", r\"\\bdudos\",\n    r\"\\bmuhtemel\\b\", r\"\\bolasi\\b\", r\"\\bsupheli\\b\", r\"\\bizlenim\",\n    r\"\\bmoguce\\b\", r\"\\bvjerojatno\\b\", r\"\\bsumnja\\b\",\n    r\"πιθαν\", r\"υποπτ\",\n    r\"\\bmoglich\", r\"\\bverdachtig\", r\"\\bfraglich\", r\"\\bV\\.a\\.\\b\",\n    r\"\\bвъзможно\\b\", r\"\\bвероятно\\b\", r\"суспект\",\n    r\"\\bmogelijk\\b\", r\"\\bverdacht\\b\",\n)\n\nTEAR = _rx(\n    r\"\\btear\", r\"\\btorn\\b\", r\"\\brupture\", r\"\\bdisruption\\b\", r\"discontinuit\",\n    r\"\\bavuls\",\n    r\"\\brotura\\b\", r\"\\broturas\\b\", r\"\\bruptura\", r\"\\bdesgarro\", r\"\\broto\\b\",\n    r\"\\bdechirure\", r\"\\bdechire\",\n    r\"\\bscheur\", r\"\\bruptuur\", r\"gescheurd\",\n    r\"\\briss\\b\", r\"einriss\", r\"\\bruptur\", r\"zerreiss\", r\"\\blasion\",\n    r\"\\byirtik\", r\"\\byirtig\", r\"\\bkopma\\b\", r\"butunluk kaybi\", r\"\\brupturu\\b\",\n    r\"\\bpuknuce\", r\"\\bruptur\", r\"\\bprekid\\b\", r\"\\bpukotin\",\n    r\"ρηξη\", r\"ρηξις\", r\"ρηγμα\",\n    r\"руптура\", r\"разкъсв\", r\"разрив\", r\"скъсв\",\n)\n\nDEGEN = _rx(\n    r\"degenerat\", r\"\\bmucoid\\b\", r\"\\bmyxoid\\b\", r\"\\bfray\", r\"\\bfissur\",\n    r\"dejeneratif\", r\"\\bmukoid\\b\", r\"degenerativn\", r\"εκφυλ\", r\"дегенерат\",\n    r\"\\bμυξοειδ\", r\"\\bμυξωδ\",\n    r\"\\bmuco ?ide\\b\", r\"aufgefasert\",\n)\n\nINJURY = _rx(\n    r\"\\binjur\", r\"\\bsprain\", r\"\\blesion\", r\"\\blasion\", r\"\\bedema\\b\", r\"\\boedema\\b\",\n    r\"\\bodem\\b\", r\"\\bedem\\b\", r\"\\bοιδημα\", r\"\\bодем\", r\"\\bедем\", r\"\\bstrain\\b\",\n    r\"\\bhigh signal\\b\", r\"\\bsignal alteration\\b\", r\"\\bhiperintens\", r\"\\bhyperintens\",\n    r\"aumento de senal\", r\"alteracion de senal\", r\"cambio de senal\",\n    r\"\\bsignalanhebung\", r\"\\bsignalalteration\", r\"verhoogd signaal\", r\"sinyal artis\",\n    r\"αυξημενο σημα\", r\"повишен сигнал\",\n    r\"\\bthicken\", r\"\\bzadebljanje\\b\", r\"\\bverdikking\\b\", r\"\\bdistenzij\",\n    r\"\\blaksite\\b\", r\"\\blaxity\\b\", r\"\\bpartial\\b\", r\"\\bparcijaln\", r\"\\bparcial\",\n    r\"\\bpartiel\", r\"\\bpartiell\",\n)\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:03.146096Z","iopub.status.busy":"2026-08-07T07:49:03.145611Z","iopub.status.idle":"2026-08-07T07:49:03.157378Z","shell.execute_reply":"2026-08-07T07:49:03.156739Z"},"papermill":{"duration":0.045802,"end_time":"2026-08-07T07:49:03.158928+00:00","exception":false,"start_time":"2026-08-07T07:49:03.113126+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-08","cell_type":"code","source":"ANAT = {\n    \"ACL\": _rx(\n        r\"anterior cruciate\", r\"\\bacl\\b\",\n        r\"cruzado anterior\", r\"\\blca\\b\",\n        r\"croise anterieur\",\n        r\"voorste kruisband\", r\"\\bvkb\\b\",\n        r\"vorderes kreuzband\", r\"vorderen kreuzband\", r\"vordere kreuzband\",\n        r\"on capraz\", r\"\\bocb\\b\",\n        r\"prednji krizni\", r\"prednjeg krizn\",\n        r\"προσθι[οα][^ ]* χιαστ\", r\"προσθιου χιαστου\", r\"χιαστο[^ ]* συνδεσμ\",\n\n        r\"\\bχιαστ\\w*\",\n        r\"предна кръстна\", r\"предната кръстна\",\n\n        r\"cruciate ligaments\", r\"ligamentos cruzados\", r\"ligaments croises\",\n        r\"kruisbanden\", r\"kreuzbander\", r\"capraz baglar\", r\"krizn[a-z]* ligament[a-z]*\",\n        r\"χιαστοι συνδεσμ\", r\"χιαστων συνδεσμ\", r\"кръстните връзки\", r\"кръстни връзки\",\n    ),\n    \"MCL\": _rx(\n        r\"medial collateral\", r\"\\bmcl\\b\", r\"tibial collateral\",\n        r\"colateral medial\", r\"colateral interno\", r\"\\blcm\\b\",\n        r\"collateral medial\", r\"collateral interne\",\n        r\"mediale collaterale\", r\"binnenband\", r\"\\b(mediale|laterale) banden\\b\",\n        r\"\\bcollaterale banden\\b\",\n        r\"innenband\", r\"mediales? kollateral\",\n        r\"\\bic yan bag\", r\"medial kollateral\", r\"\\biyb\\b\",\n        r\"medijalni kolateraln\", r\"medijalnog kolateraln\",\n        r\"εσω πλαγι\", r\"εσωτερικο πλαγι\", r\"\\bπλαγι\\w* συνδεσμ\", r\"\\bπλαγιοι\\b\",\n        r\"медиален колатерал\", r\"вътрешна странична\", r\"\\bколатерал\\w*\",\n\n        r\"\\bcolaterales\\b\", r\"\\bcollateraux\\b\", r\"\\bcollateralen\\b\", r\"\\bkolateralni\\b\",\n        r\"collateral ligaments\", r\"ligamentos colaterales\", r\"ligaments collateraux\",\n        r\"collaterale banden\", r\"kollateralbander\", r\"seitenbander\", r\"yan baglar\",\n        r\"kolateraln[a-z]* ligament[a-z]*\", r\"πλαγιοι συνδεσμ\", r\"πλαγιων συνδεσμ\",\n        r\"колатерални връзки\", r\"страничните връзки\",\n    ),\n    \"Medial Meniscus\": _rx(\n        r\"medial meniscus\", r\"\\bmm\\b(?= tear)\", r\"medial menisc\",\n        r\"menisco medial\", r\"menisco interno\",\n        r\"menisque medial\", r\"menisque interne\",\n        r\"mediale meniscus\", r\"binnenmeniscus\",\n        r\"innenmeniskus\", r\"medialen? meniskus\", r\"innenmeniskushinterhorn\",\n        r\"medyal menisk\", r\"\\bic menisk\",\n        r\"medijalni meniskus\", r\"medijalnog meniskusa\", r\"medijalnom meniskusu\",\n        r\"εσω μηνισκ\", r\"μηνισκ[^ ]* του εσω\", r\"εσω διαμερισμα[^.]{0,40}μηνισκ\",\n        r\"медиалния менискус\", r\"медиален менискус\", r\"вътрешния менискус\",\n    ),\n    \"Lateral Meniscus\": _rx(\n        r\"lateral meniscus\", r\"lateral menisc\",\n        r\"menisco lateral\", r\"menisco externo\",\n        r\"menisque lateral\", r\"menisque externe\",\n        r\"laterale meniscus\", r\"buitenmeniscus\",\n        r\"aussenmeniskus\", r\"lateralen? meniskus\",\n        r\"lateral menisk\", r\"\\bdis menisk\",\n        r\"lateralni meniskus\", r\"lateralnog meniskusa\", r\"lateralnom meniskusu\",\n        r\"εξω μηνισκ\", r\"μηνισκ[^ ]* του εξω\", r\"εξω διαμερισμα[^.]{0,40}μηνισκ\",\n        r\"латералния менискус\", r\"латерален менискус\", r\"външния менискус\",\n    ),\n}\n\nOA_EVIDENCE = _rx(\n    r\"osteoarthrit\", r\"\\barthros\", r\"\\bgonarthros\", r\"\\bosteoarthros\",\n    r\"chondropath\", r\"chondromalac\", r\"condropat\", r\"condromalac\",\n    r\"cartilage loss\", r\"cartilage thinning\", r\"chondral (loss|defect|ulcer|thinning)\",\n    r\"osteophyt\", r\"osteofit\", r\"osteofyt\", r\"osteofito\", r\"osteophyten\",\n    r\"joint space narrowing\", r\"pinzamiento articular\",\n    r\"kikirdak kayb\", r\"kikirdak incelme\", r\"kondropati\", r\"kondral\",\n    r\"kraakbeen(lijden|verlies)\", r\"gonartrose\", r\"artrose\",\n    r\"knorpel(verlust|schaden|defekt)\", r\"arthrose\", r\"gonarthrose\",\n    r\"hrskavic\", r\"hondromalac\", r\"artroz\", r\"osteoartrit\",\n    r\"χονδρ[^ ]*παθ\", r\"αρθριτ\", r\"αρθρωσ\", r\"οστεοφυτ\",\n    r\"αρθρικου χονδρου\", r\"εξαλειψη του αρθρικου χονδρου\",\n    r\"артроз\", r\"хондропат\", r\"остеофит\", r\"хрущял[^.]{0,30}(изтън|увред|дефект)\",\n    r\"ulcera[s]? condral\", r\"cartilago[^.]{0,25}(perdida|adelgaz)\",\n    r\"icrs grade\", r\"outerbridge\",\n)\n\nCOMPARTMENT = {\n    \"Medial OA\": _rx(\n        r\"medial (femorotibial|tibiofemoral|compartment)\",\n        r\"compartimento femorotibial medial\", r\"femorotibial interno\",\n        r\"mediaal femorotibiaal\", r\"mediale femorotibial\",\n        r\"medial femorotibial\", r\"medialen kompartiment\", r\"innere[sn]? kompartiment\",\n        r\"medyal femorotibial\", r\"ic kompartman\", r\"medyal kompartman\",\n        r\"medijaln[^ ]* (femorotibi|odjelj|kompartm)\",\n        r\"εσω διαμερισμα\", r\"εσω κνημιαι\", r\"εσω μηριαι\",\n        r\"медиалн[^ ]* (компартм|отдел|тибиал|феморотиб)\",\n        r\"medial (femoral|tibial) (condyle|plateau)\", r\"condilo femoral medial\",\n        r\"medialen? (femurkondyl|tibiaplateau)\", r\"mediale femorale condyl\",\n    ),\n    \"Lateral OA\": _rx(\n        r\"lateral (femorotibial|tibiofemoral|compartment)\",\n        r\"compartimento femorotibial lateral\", r\"femorotibial externo\",\n        r\"lateraal femorotibiaal\", r\"laterale femorotibial\",\n        r\"lateral femorotibial\", r\"lateralen kompartiment\", r\"aussere[sn]? kompartiment\",\n        r\"lateral femorotibial\", r\"dis kompartman\", r\"lateral kompartman\",\n        r\"lateraln[^ ]* (femorotibi|odjelj|kompartm)\",\n        r\"εξω διαμερισμα\", r\"εξω κνημιαι\", r\"εξω μηριαι\",\n        r\"латералн[^ ]* (компартм|отдел|тибиал|феморотиб)\",\n        r\"lateral (femoral|tibial) (condyle|plateau)\", r\"condilo femoral lateral\",\n        r\"lateralen? (femurkondyl|tibiaplateau)\", r\"laterale femorale condyl\",\n    ),\n    \"PF OA\": _rx(\n        r\"patellofemoral\", r\"femoropatellar\", r\"femoropatelar\", r\"patelofemoral\",\n        r\"retropatellar\", r\"retrorotulian\", r\"\\btrochlea\", r\"\\btroclea\", r\"\\btroklea\",\n        r\"\\bpatella\\b\", r\"\\bpatellar\\b\", r\"\\brotulian\", r\"\\brotula\\b\", r\"\\bpatele\\b\",\n        r\"\\bpatellae?\\b\", r\"patellofemoraal\", r\"femoropatellair\",\n        r\"επιγονατιδ\", r\"μηροεπιγονατιδ\", r\"τροχιλ\",\n        r\"пател\", r\"феморопател\", r\"тролх\",\n        r\"anterior compartment\", r\"compartimento anterior\", r\"prednj[^ ]* odjeljk\",\n    ),\n}\n\nDIRECT = {\n    \"Effusion\": _rx(\n        r\"\\beffusion\", r\"joint fluid\", r\"intra ?articular fluid\", r\"\\bhydrops\\b\",\n        r\"derrame articular\", r\"\\bderrame\\b\", r\"liquido articular\",\n        r\"epanchement\",\n        r\"gewrichtsvocht\", r\"\\bvocht\\b\", r\"\\bhydrops\\b\", r\"gewrichtseffusie\",\n        r\"gelenkerguss\", r\"\\berguss\\b\", r\"gelenksergu\",\n\n        r\"eklem\\w* ic\\w* sivi\", r\"efuzyon\", r\"eklem sivisi\",\n        r\"sivi (miktari|artisi|birikimi)\", r\"sivi artis\", r\"\\bsivi\\b[^.]{0,25}artmis\",\n        r\"\\bizljev\", r\"\\bizliv\", r\"zglobn[^ ]* tekucin\", r\"\\bhidrops\\b\",\n        r\"αρθρικ[^ ]* υγρ\", r\"υγρου ενδαρθρικα\", r\"ενδαρθρικ[^ ]* υγρ\", r\"ποσοτητα υγρου\",\n        r\"ενδαρθρικ\", r\"αρθρικη συλλογη\", r\"υγρο στην αρθρωση\", r\"υγρου στην αρθρωση\",\n        r\"ставен излив\", r\"излив\", r\"ставна течност\", r\"синовиална течност\",\n    ),\n    \"Synovitis\": _rx(\n        r\"synovit\", r\"sinovit\", r\"synovial (thickening|proliferation|hypertroph)\",\n        r\"synovitis\", r\"synoviale? (verdikking|proliferatie)\",\n        r\"synovialitis\", r\"synovialis(verdickung|proliferation)\",\n        r\"sinovijalitis\", r\"sinovitis\", r\"zadebljanje sinovij\",\n        r\"υμενιτιδα\", r\"συνοβιτιδα\", r\"υμενικ[^ ]* υπερτροφ\", r\"αρθρικου υμεν\",\n        r\"синовит\", r\"синовиал[^ ]* (задебел|пролифер)\",\n        r\"verdikkingen van (het )?synovium\", r\"pannus\",\n    ),\n    \"Baker's\": _rx(\n        r\"baker\", r\"popliteal cyst\", r\"quiste popliteo\", r\"quistes popliteos\",\n        r\"kyste poplite\", r\"popliteale? cyst\", r\"poplitealzyste\", r\"bakerzyste\",\n        r\"popliteal kist\", r\"\\bbakerova\\b\", r\"poplitealn[^ ]* cist\",\n        r\"κυστη baker\", r\"πολυχωρη συνοβιακη κυστη\", r\"κυστη του baker\",\n        r\"киста на бейкър\", r\"бейкърова киста\", r\"поплитеална киста\",\n        r\"gastrocnemio ?semimembranos\", r\"gastrocnemius semimembranosus burs\",\n    ),\n    \"Contusion\": _rx(\n        r\"\\bcontusion\", r\"bone bruise\", r\"bone marrow (o?edema|contusion)\",\n        r\"\\bkontuz\", r\"medular bone o?edema\", r\"marrow o?edema\",\n        r\"contusion osea\", r\"edema oseo\", r\"edema de medula osea\",\n        r\"oedeme osseux\", r\"contusion osseuse\",\n        r\"botcontusie\", r\"botoedeem\", r\"beenmergoedeem\", r\"botmergoedeem\",\n        r\"knochenmarkodem\", r\"knochenodem\", r\"kontusion\", r\"bone bruise\",\n        r\"kemik kontuzyonu\", r\"kemik iligi odemi\", r\"kemik odemi\",\n        r\"kostani edem\", r\"edem kosti\", r\"kontuzij\",\n        r\"οστεομυελικ[^ ]* οιδημα\", r\"οστικο οιδημα\", r\"μυελικο οιδημα\",\n        r\"костномозъчен едем\", r\"костен едем\", r\"контузионен\",\n    ),\n    \"Fracture\": _rx(\n        r\"\\bfractur\", r\"\\bfract\\b\",\n        r\"\\bfractura\", r\"\\bfracturas\\b\",\n        r\"\\bfractuur\", r\"\\bbreuk\\b\",\n        r\"\\bfraktur\", r\"\\bbruch\\b\",\n        r\"\\bkirik\\b\", r\"\\bkirigi\\b\", r\"\\bkirik\\b\",\n        r\"\\bfraktur\", r\"\\bprijelom\", r\"impresijsk[^ ]* fraktur\",\n        r\"καταγμα\", r\"καταγματ\",\n        r\"фрактур\", r\"счупван\", r\"фисур\",\n        r\"insufficiency fracture\", r\"stress fracture\", r\"avulsion fracture\",\n        r\"subchondral fracture\", r\"subkondral kiri\",\n    ),\n}\n\nDECOY = {\n\n    \"Fracture\": _rx(r\"microfractur\", r\"\\bfracture (risk|prophyla)\"),\n    \"Baker's\": _rx(r\"meniscal cyst\", r\"quiste meniscal\", r\"ganglion\"),\n}\n\nPAIRED = {\"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\"}\nOA_TARGETS = {\"Medial OA\", \"Lateral OA\", \"PF OA\"}\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:03.222516Z","iopub.status.busy":"2026-08-07T07:49:03.222115Z","iopub.status.idle":"2026-08-07T07:49:03.246259Z","shell.execute_reply":"2026-08-07T07:49:03.245483Z"},"papermill":{"duration":0.058242,"end_time":"2026-08-07T07:49:03.247632+00:00","exception":false,"start_time":"2026-08-07T07:49:03.18939+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-09","cell_type":"code","source":"STEM_MENISCUS = _rx(r\"menisc\\w*\", r\"menisk\\w*\", r\"μηνισκ\\w*\", r\"мениск\\w*\")\nSTEM_CRUCIATE = _rx(r\"cruciate\", r\"cruzado\", r\"croise\", r\"kruisband\", r\"kreuzband\",\n                    r\"capraz bag\\w*\", r\"krizn\\w*\", r\"χιαστ\\w*\", r\"кръстн\\w*\",\n                    r\"\\bacl\\b\", r\"\\bpcl\\b\", r\"\\blca\\b\", r\"\\blcp\\b\", r\"\\bvkb\\b\",\n                    r\"\\bhkb\\b\", r\"\\bocb\\b\", r\"\\bacb\\b\")\nSTEM_COLLATERAL = _rx(r\"collateral\\w*\", r\"colateral\\w*\", r\"kollateral\\w*\",\n                      r\"collaterale\\w*\", r\"kolateraln\\w*\", r\"yan bag\\w*\",\n                      r\"πλαγι\\w*\", r\"колатерал\\w*\", r\"странич\\w*\",\n                      r\"innenband\\w*\", r\"aussenband\\w*\", r\"binnenband\\w*\",\n                      r\"\\bmcl\\b\", r\"\\blcl\\b\", r\"\\blcm\\b\", r\"\\biyb\\b\")\n\nSIDE_MEDIAL = _rx(r\"\\bmedial\\w*\", r\"\\bmedyal\\w*\", r\"\\bmedijaln\\w*\", r\"\\bmediaal\\w*\",\n                  r\"\\bmediale\\w*\", r\"\\bintern[oa]\\w*\", r\"\\binterne\\w*\", r\"\\binnen\\w*\",\n                  r\"\\bic\\b\", r\"\\bunutarnj\\w*\", r\"\\bεσω\\w*\", r\"\\bεσωτερικ\\w*\",\n                  r\"\\bмедиал\\w*\", r\"\\bвътреш\\w*\", r\"\\btibial collateral\\b\",\n                  r\"\\bbinnen\\w*\", r\"\\bmediaal\\b\")\nSIDE_LATERAL = _rx(r\"\\blateral\\w*\", r\"\\bextern[oa]\\w*\", r\"\\bexterne\\w*\", r\"\\bdis\\b\",\n                   r\"\\blateraln\\w*\", r\"\\baussen\\w*\", r\"\\bbuiten\\w*\", r\"\\bεξω\\w*\",\n                   r\"\\bεξωτερικ\\w*\", r\"\\bлатерал\\w*\", r\"\\bвъншн\\w*\",\n                   r\"\\bfibular collateral\\b\", r\"\\bvanjsk\\w*\")\nSIDE_ANTERIOR = _rx(r\"\\banterior\\w*\", r\"\\bant\\b\", r\"\\bon\\b\", r\"\\bprednj\\w*\",\n                    r\"\\bvorder\\w*\", r\"\\bvoorste\\b\", r\"\\bπροσθι\\w*\", r\"\\bпредн\\w*\",\n                    r\"\\banteriyor\\w*\", r\"\\bavant\\b\", r\"\\bant[eé]rieur\\w*\")\n\nSIDE_POSTERIOR = _rx(r\"\\bposterior\\w*\", r\"\\bpost[eé]rieur\\w*\", r\"\\bposteriore\\w*\",\n                     r\"\\bhinter\\w*\", r\"\\bachterste\\b\", r\"\\barka\\b\", r\"\\bstraznj\\w*\",\n                     r\"\\bzadnj\\w*\", r\"\\bοπισθι\\w*\", r\"\\bзадн\\w*\", r\"\\bpostero\\w*\")\n\nSTEM_FRACTURE = _rx(r\"fractur\\w*\", r\"fraktur\\w*\", r\"fractuur\\w*\", r\"\\bfract\\b\",\n                    r\"kiri[kgğ]\\w*\", r\"prijelom\\w*\", r\"lom kosti\", r\"\\bbreuk\\w*\",\n                    r\"\\bbruch\\w*\", r\"καταγμα\\w*\", r\"καταγματ\\w*\", r\"фрактур\\w*\",\n                    r\"счупван\\w*\", r\"fisur\\w* (osea|oseas|kost)\", r\"fissur\\w* kost\")\n\nSTEM_OA_COMPARTMENT = _rx(r\"compartment\\w*\", r\"compartimento\\w*\", r\"compartiment\\w*\",\n                          r\"kompartman\\w*\", r\"kompartiment\\w*\", r\"odjelj\\w*\",\n                          r\"διαμερισμα\\w*\", r\"компартм\\w*\", r\"\\bотдел\\w*\",\n                          r\"femorotibial\\w*\", r\"femorotibiaal\\w*\", r\"tibiofemoral\\w*\",\n                          r\"femoro tibial\\w*\", r\"κνημιαι\\w*\", r\"μηριαι\\w*\",\n                          r\"femoral condyl\\w*\", r\"tibial plateau\\w*\",\n                          r\"condilo femoral\", r\"platillo tibial\", r\"tibiaplateau\\w*\",\n                          r\"femurkondyl\\w*\", r\"femoralne? kondil\\w*\",\n                          r\"tibijaln\\w* plato\", r\"femoral kondil\\w*\",\n                          r\"tibia plato\", r\"tibyal plato\")\n\n\ndef _near(clause: str, stem_rx: re.Pattern, qual_rx: re.Pattern, window: int = 55):\n    \"\"\"True if a stem match has a qualifier within `window` characters either side.\n\n    Character windows rather than token windows, because word order differs: English\n    puts the side before the noun, Greek and Bulgarian often after, and Turkish\n    attaches it as a separate preceding adjective.\n    \"\"\"\n    for m in stem_rx.finditer(clause):\n        lo = max(0, m.start() - window)\n        hi = min(len(clause), m.end() + window)\n        if qual_rx.search(clause[lo:hi]):\n            return True\n    return False\n\n\nSTEM_RULES = {\n    \"ACL\": (STEM_CRUCIATE, SIDE_ANTERIOR),\n    \"MCL\": (STEM_COLLATERAL, SIDE_MEDIAL),\n    \"Medial Meniscus\": (STEM_MENISCUS, SIDE_MEDIAL),\n    \"Lateral Meniscus\": (STEM_MENISCUS, SIDE_LATERAL),\n    \"Medial OA\": (STEM_OA_COMPARTMENT, SIDE_MEDIAL),\n    \"Lateral OA\": (STEM_OA_COMPARTMENT, SIDE_LATERAL),\n}\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:03.308288Z","iopub.status.busy":"2026-08-07T07:49:03.30786Z","iopub.status.idle":"2026-08-07T07:49:03.321185Z","shell.execute_reply":"2026-08-07T07:49:03.320313Z"},"papermill":{"duration":0.046074,"end_time":"2026-08-07T07:49:03.322707+00:00","exception":false,"start_time":"2026-08-07T07:49:03.276633+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-10","cell_type":"code","source":"SEV_LOW = _rx(\n    r\"\\bsmall\\b\", r\"\\bminimal\\b\", r\"\\btrace\\b\", r\"\\bmild\\b\", r\"\\bslight\\b\",\n    r\"\\btiny\\b\", r\"\\bscant\\b\", r\"\\bmimimal\\b\", r\"\\bdiscrete\\b\", r\"\\bfocal\\b\",\n    r\"\\bleve\\b\", r\"\\bminim\", r\"\\bpeque\", r\"\\bligero\\b\", r\"\\bescaso\\b\", r\"\\bdiscreto\\b\",\n    r\"\\bhafif\\b\", r\"\\bminimal\\b\", r\"\\baz miktarda\\b\", r\"\\bsilik\\b\",\n    r\"\\bmanja\\b\", r\"\\bmanji\\b\", r\"\\bblago\\b\", r\"\\bdiskretn\", r\"\\bmalo\\b\",\n    r\"\\bgering\", r\"\\bdiskret\", r\"\\bkleine?r?\\b\", r\"\\bwenig\\b\", r\"\\bzarte?\\b\",\n    r\"\\bbeperkte?\\b\", r\"\\bgeringe\\b\", r\"\\bweinig\\b\", r\"\\blichte?\\b\",\n    r\"\\bηπι\", r\"\\bμικρ\", r\"\\bελαχιστ\",\n    r\"\\bминимал\", r\"\\bлек\", r\"\\bмалк\", r\"\\bнеголям\",\n)\n\nSEV_HIGH = _rx(\n    r\"\\blarge\\b\", r\"\\bmarked\\b\", r\"\\bmassive\\b\", r\"\\bsevere\\b\", r\"\\bextensive\\b\",\n    r\"\\bmoderate\\b\", r\"\\bgross\\b\", r\"\\bsignificant\\b\", r\"\\babundant\\b\", r\"\\btense\\b\",\n    r\"\\bmoderad\", r\"\\bimportante\\b\", r\"\\bsevera?\\b\", r\"\\bmarcad\", r\"\\bcuantios\",\n    r\"\\bbelirgin\\b\", r\"\\byaygin\\b\", r\"\\bileri\\b\", r\"\\bciddi\\b\", r\"\\bbol\\b\",\n    r\"\\bopsezan\\b\", r\"\\bveliki\\b\", r\"\\bizrazit\", r\"\\bznacajn\", r\"\\bumjeren\",\n    r\"\\bausgepragt\", r\"\\bdeutlich\", r\"\\bmassiv\", r\"\\bmassig\", r\"\\bgross\",\n    r\"\\buitgebreid\", r\"\\bgevorderd\", r\"\\bveel\\b\", r\"\\bmatige?\\b\",\n    r\"\\bμετρι\", r\"\\bμεγαλ\", r\"\\bεκτεταμεν\", r\"\\bευμεγεθ\", r\"\\bσοβαρ\",\n    r\"\\bголям\", r\"\\bизразен\", r\"\\bзначим\", r\"\\bумерен\", r\"\\bобилен\",\n)\n\n\nGLOBAL_OA = _rx(\n    r\"tri ?compartment\", r\"all three compartment\", r\"global(ised)? (oa|osteoarthrit)\",\n    r\"\\bgonarthros\", r\"\\bgonartros\", r\"\\bgonarthrose\", r\"\\bgonartrose\",\n    r\"osteoarthritis of the knee\", r\"artrosis (de |)(la )?rodilla\", r\"knee osteoarthrit\",\n    r\"\\bdiz osteoartrit\", r\"\\bgonartroz\", r\"artroza koljena\",\n    r\"οστεοαρθριτιδα\", r\"αρθριτιδα του γονατος\",\n    r\"артроза на колянната\", r\"гонартроз\",\n    r\"degenerative joint disease\", r\"\\bdjd\\b\",\n)\n\nDEGENERATIVE_MARROW = _rx(\n    r\"subchondral\", r\"subcondral\", r\"subkondral\", r\"supkondraln\", r\"subchondraln\",\n    r\"υποχονδρι\", r\"субхондрал\", r\"subchondrale?\",\n    r\"\\bcyst\", r\"\\bquist\", r\"\\bzyste\\b\", r\"\\bcistic\", r\"reactive\", r\"reactivo\",\n)\n\nTRAUMA = _rx(\n    r\"\\bbruise\\b\", r\"\\bcontusion\", r\"\\bkontuz\", r\"\\bcontusion osea\\b\",\n    r\"\\btrauma\", r\"\\bimpaction\\b\", r\"\\bpivot shift\\b\", r\"\\bkissing\\b\",\n    r\"\\bacute\\b\", r\"\\bagudo\\b\", r\"\\bakut\", r\"\\bpivot kaymasi\\b\",\n    r\"\\bcontusion osseuse\\b\", r\"\\bbone bruise\\b\", r\"\\bbotcontusie\\b\",\n    r\"\\bконтузион\", r\"\\bμωλωπ\", r\"\\bkontuzij\",\n)\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:03.38372Z","iopub.status.busy":"2026-08-07T07:49:03.383442Z","iopub.status.idle":"2026-08-07T07:49:03.392893Z","shell.execute_reply":"2026-08-07T07:49:03.39211Z"},"papermill":{"duration":0.042349,"end_time":"2026-08-07T07:49:03.394452+00:00","exception":false,"start_time":"2026-08-07T07:49:03.352103+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-11","cell_type":"code","source":"def _polarity(clause: str, anchor_end: int) -> str:\n    \"\"\"Classify one clause as positive, negative or uncertain for a matched term.\n\n    Scope is the whole clause. Clause segmentation already keeps statements short, and\n    a window in characters mis-scopes badly across languages with different word orders -\n    Turkish puts its negator at the end of the sentence, English at the front.\n    \"\"\"\n    if UNCERTAIN.search(clause):\n        return \"uncertain\"\n    if NEGATION.search(clause):\n        return \"negative\"\n    if NORMALITY.search(clause):\n        if TEAR.search(clause) or re.search(r\"\\bgrade [34]\\b\", clause):\n            return \"positive\"\n        return \"negative\"\n    return \"positive\"\n\n\nclass _Matcher:\n    \"\"\"Phrase lexicon first, stem+side proximity as the fallback.\n\n    Exposes `.search` so it drops into the same slot as a compiled pattern.\n    \"\"\"\n\n    def __init__(self, phrase_rx, stem=None, side=None, window=55, contrary=None):\n        self.phrase_rx = phrase_rx\n        self.stem = stem\n        self.side = side\n        self.window = window\n        self.contrary = contrary\n\n    def search(self, clause):\n        m = self.phrase_rx.search(clause)\n        if m is not None and not self._wrong_side(clause):\n            return m\n        if self.stem is not None and _near(clause, self.stem, self.side, self.window):\n            return self.stem.search(clause)\n        return None\n\n    def _wrong_side(self, clause):\n        \"\"\"True when the clause names the other member of this structure's pair.\n\n        Some cues in the lexicon are side-blind by design: Greek separates the adjective\n        from its noun (\"cruciate and collateral ligaments\"), so the bare adjective stem\n        has to stand alone or the clause is lost. That stem then also matches the\n        posterior cruciate and the lateral collateral, neither of which is a target here,\n        and a positive outranks every negative in the scorer - so one PCL clause was\n        enough to override an explicit \"the ACL is normal\".\n\n        The test is proximity to the structure's own stem, not presence in the clause.\n        \"Posterior horn of the medial meniscus\" appears in a large share of knee reports\n        and says nothing about a cruciate; only a qualifier sitting beside the ligament\n        word is one. A clause naming both sides keeps the match, because it does mention\n        this target.\n        \"\"\"\n        if self.contrary is None or self.stem is None:\n            return False\n        return (_near(clause, self.stem, self.contrary, self.window)\n                and not _near(clause, self.stem, self.side, self.window))\n\n\nCONTRARY = {\"ACL\": SIDE_POSTERIOR, \"MCL\": SIDE_LATERAL}\n\nANAT_MATCH = {\n    tgt: _Matcher(ANAT[tgt], *STEM_RULES[tgt], contrary=CONTRARY.get(tgt))\n    for tgt in PAIRED\n}\nCOMPARTMENT_MATCH = {\n    \"Medial OA\": _Matcher(COMPARTMENT[\"Medial OA\"], *STEM_RULES[\"Medial OA\"]),\n    \"Lateral OA\": _Matcher(COMPARTMENT[\"Lateral OA\"], *STEM_RULES[\"Lateral OA\"]),\n    \"PF OA\": _Matcher(COMPARTMENT[\"PF OA\"]),\n}\nDIRECT_MATCH = {\n    tgt: _Matcher(_rx(rx.pattern, STEM_FRACTURE.pattern) if tgt == \"Fracture\" else rx)\n    for tgt, rx in DIRECT.items()\n}\n\n\ndef _severity(clause: str) -> float:\n    \"\"\"Weight one positive mention by how emphatic the sentence is.\n\n    Ordered, not calibrated. A \"moderate effusion\" must outrank a \"trace effusion\" and\n    both must outrank silence; the absolute numbers do not matter to AUC.\n    \"\"\"\n    high = SEV_HIGH.search(clause) is not None\n    low = SEV_LOW.search(clause) is not None\n    if high and not low:\n        return 1.0\n    if low and not high:\n        return 0.45\n    return 0.75                       # unqualified mention\n\n\ndef _score_clauses(cls, anat_rx, path_rx=None, decoy_rx=None, context_penalty=None,\n                   context_bonus=None):\n    \"\"\"Accumulate graded evidence over clauses for one target.\n\n    Returns (score, confidence, n_pos, n_neg). Positives are graded by severity and by\n    optional context regexes; negatives only matter when nothing positive was found,\n    because reports assert normality for every structure they check.\n    \"\"\"\n    n_pos = n_neg = n_unc = 0\n    best = 0.0\n    for c in cls:\n        m = anat_rx.search(c)\n        if not m:\n            continue\n        if decoy_rx is not None and decoy_rx.search(c):\n            continue\n        if path_rx is not None and not path_rx.search(c):\n            if NORMALITY.search(c) and not NEGATION.search(c):\n                n_neg += 1\n            continue\n        pol = _polarity(c, m.end())\n        if pol == \"positive\":\n            n_pos += 1\n            w = _severity(c)\n            if context_penalty is not None and context_penalty.search(c):\n                w *= 0.45\n            if context_bonus is not None and context_bonus.search(c):\n                w = min(1.0, w * 1.35)\n            best = max(best, w)\n        elif pol == \"negative\":\n            n_neg += 1\n        else:\n            n_unc += 1\n            best = max(best, 0.30)\n\n    if n_pos or n_unc:\n        \n        score = min(0.95, 0.50 + 0.42 * best + 0.03 * min(n_pos, 3))\n        conf = min(1.0, 0.55 + 0.15 * n_pos)\n    elif n_neg:\n        score = max(0.04, 0.20 - 0.04 * n_neg)\n        conf = min(0.9, 0.45 + 0.12 * n_neg)\n    else:\n        score, conf = 0.28, 0.05          \n    return score, conf, n_pos, n_neg\n\n\ndef extract(report: str) -> dict:\n    \"\"\"Extract twelve (score, confidence) pairs from one report.\"\"\"\n    cls = clauses(report)\n    out = {}\n    path_paired = _rx(TEAR.pattern, DEGEN.pattern, INJURY.pattern)\n\n    for tgt in TARGETS:\n        if tgt in PAIRED:\n            s, c, npos, nneg = _score_clauses(cls, ANAT_MATCH[tgt], path_paired)\n        elif tgt in OA_TARGETS:\n            s, c, npos, nneg = _score_clauses(cls, COMPARTMENT_MATCH[tgt], OA_EVIDENCE)\n        elif tgt == \"Contusion\":\n\n            s, c, npos, nneg = _score_clauses(cls, DIRECT_MATCH[tgt], None, DECOY.get(tgt),\n                                              context_penalty=DEGENERATIVE_MARROW,\n                                              context_bonus=TRAUMA)\n        else:\n            s, c, npos, nneg = _score_clauses(cls, DIRECT_MATCH[tgt], None, DECOY.get(tgt))\n        out[tgt] = s\n        out[tgt + \"__conf\"] = c\n        out[tgt + \"__npos\"] = npos\n        out[tgt + \"__nneg\"] = nneg\n\n\n    g_hits = [c for c in cls if GLOBAL_OA.search(c) and _polarity(c, 0) == \"positive\"]\n    if g_hits:\n        gscore = 0.50 + 0.42 * max(_severity(c) for c in g_hits)\n        for tgt in OA_TARGETS:\n            if out[tgt + \"__npos\"] == 0 and out[tgt + \"__nneg\"] == 0:\n                out[tgt] = max(out[tgt], gscore * 0.92)\n                out[tgt + \"__conf\"] = max(out[tgt + \"__conf\"], 0.4)\n\n    if out[\"Synovitis__npos\"] == 0 and out[\"Synovitis__nneg\"] == 0:\n        out[\"Synovitis\"] = max(out[\"Synovitis\"], 0.28 + 0.45 * (out[\"Effusion\"] - 0.28))\n\n    return out\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:03.457329Z","iopub.status.busy":"2026-08-07T07:49:03.45686Z","iopub.status.idle":"2026-08-07T07:49:03.476614Z","shell.execute_reply":"2026-08-07T07:49:03.476084Z"},"papermill":{"duration":0.053457,"end_time":"2026-08-07T07:49:03.478058+00:00","exception":false,"start_time":"2026-08-07T07:49:03.424601+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-13","cell_type":"code","source":"from __future__ import annotations\n\nimport os\n\nfor _v in (\"OMP_NUM_THREADS\", \"OPENBLAS_NUM_THREADS\", \"MKL_NUM_THREADS\"):\n    os.environ.setdefault(_v, \"4\")\n\nimport gc\nimport re\nimport time\nimport traceback\nfrom concurrent.futures import ThreadPoolExecutor\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\n\n\nT0 = time.time()\nSEED = 2026\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\n\nTARGETS = [\"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\", \"Medial OA\",\n           \"Lateral OA\", \"PF OA\", \"Effusion\", \"Synovitis\", \"Baker's\",\n           \"Contusion\", \"Fracture\"]\n\n\n\nCROP_MM = 130.0\n\nCACHE_IMG = 336\nGROUP = 3                 \nN_GROUP_MAX = 1\nCACHE_FRACTION = 0.45     \nCACHE_BUDGET_MAX_GB = 24.0 \nCACHE_BUDGET_GB = 12.0     \nTEST_SHARE = 0.30       \n                        \nHDR_THREADS = 16\nPIX_THREADS = 12\nORDER_THREADS = 32      \nORDER_BUDGET_S = 5400\n\nRUNS = [\n    {\"name\": \"r224\", \"img\": 224},\n    {\"name\": \"r336\", \"img\": 336},\n]\n\nEPOCHS = 10\nBATCH_STUDIES = 8          \nAUG_ROT_DEG = 8.0          \nAUG_SCALE = 0.08\nAUG_SHIFT = 0.05\nAUG_INTENSITY = 0.10\nLAT_MIN_OFFSET_MM = 20.0   \n                           \nSLICE_BAND = (0.20, 0.80)  \nLR_HEAD = 1e-3\nLR_BACKBONE = 8e-6         \nUNFREEZE_LAST = 6          \nWEIGHT_DECAY = 0.02\nEVAL_BATCH = 8\nTIME_BUDGET = 8.0 * 3600\n\n\nSLOTS_RECOVERED = [\n    (\"SAG_FLUID_FS\", \"Sagittal\", True, True),\n    (\"COR_FLUID_FS\", \"Coronal\", True, True),\n    (\"AX_FLUID_FS\", \"Axial\", True, True),\n    (\"SAG_FLUID_NOFS\", \"Sagittal\", True, False),\n    (\"COR_T1\", \"Coronal\", False, False),\n    (\"SAG_T1\", \"Sagittal\", False, False),\n]\n\nSLOTS_PUBLIC = [\n    (\"SAG_FLUID\", \"Sagittal\", None, True),\n    (\"COR_FLUID\", \"Coronal\", None, True),\n    (\"AX_FLUID\", \"Axial\", None, True),\n    (\"SAG_STRUCT\", \"Sagittal\", None, False),\n    (\"COR_STRUCT\", \"Coronal\", None, False),\n    (\"AX_STRUCT\", \"Axial\", None, False),\n]\n\nSLOT_SCHEME = os.environ.get(\"SLOT_SCHEME\", \"recovered\")\nSLOTS = SLOTS_PUBLIC if SLOT_SCHEME == \"public\" else SLOTS_RECOVERED\nN_SLOT = len(SLOTS)\n\nFATSAT_OPTS = {\"FS\", \"FATSAT\", \"FAT_SAT\", \"FSAT\"}\n_SEP = re.compile(r\"[_\\-.]\")\n_FATSAT_RX = re.compile(r\"\\bfs\\b|fatsat|fat sat|\\bstir\\b|\\bspair\\b|\\bspir\\b|\\bwe\\b|\"\n                        r\"water excit|\\btirm\\b|\\bsting\\b|\\bfatsup\\b\")\n_T1_RX = re.compile(r\"\\bt1\\b|\\bt1w\\b\")\n_T2_RX = re.compile(r\"\\bt2\\b|\\bt2w\\b\")\n_PD_RX = re.compile(r\"\\bpd\\b|\\bpdw\\b|proton|\\bdp\\b|dens\")\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:03.602564Z","iopub.status.busy":"2026-08-07T07:49:03.601994Z","iopub.status.idle":"2026-08-07T07:49:11.456657Z","shell.execute_reply":"2026-08-07T07:49:11.455972Z"},"papermill":{"duration":7.889006,"end_time":"2026-08-07T07:49:11.458594+00:00","exception":false,"start_time":"2026-08-07T07:49:03.569588+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-14","cell_type":"code","source":"def log(msg):\n    print(f\"[{time.time() - T0:7.1f}s] {msg}\", flush=True)\n\n\ndef find_root():\n    for c in [Path(\"/kaggle/input/competitions/rsna-knee-abnormality-detection\"),\n              Path(\"/kaggle/input/rsna-knee-abnormality-detection\"),\n              Path(\"data\"), Path(\".\")]:\n        if (c / \"test.csv\").is_file() and (c / \"test_series\").is_dir():\n            return c\n    base = Path(\"/kaggle/input\")\n    if base.is_dir():\n        for depth1 in sorted(p for p in base.iterdir() if p.is_dir()):\n            for cand in [depth1] + sorted(p for p in depth1.iterdir() if p.is_dir()):\n                if (cand / \"test.csv\").is_file():\n                    return cand\n    raise FileNotFoundError(\n        f\"competition mount not found (cwd {Path.cwd()}); expected a directory holding \"\n        f\"test.csv and test_series/\")\n\n\ndef find_dinov2(variant=\"small\"):\n    \"\"\"Locate a mounted DINOv2 checkpoint directory by variant name.\"\"\"\n    base = Path(\"/kaggle/input\")\n    if not base.is_dir():\n        return None\n    hits = []\n    for root, dirs, files in os.walk(base):\n        dirs[:] = [d for d in dirs if d not in (\"train_series\", \"test_series\")]\n        if \"config.json\" in files and \"dinov2\" in root.lower():\n            hits.append(Path(root))\n    for h in hits:\n        if variant in str(h).lower():\n            return h\n    return hits[0] if hits else None\n\n\nLABEL_COLS = TARGETS + [t + \"__conf\" for t in TARGETS]\n\n\nclass LabelSourceError(RuntimeError):\n    \"\"\"Raised when the labels did not come from where this run intended.\n\n    Every other failure in this file is better survived than reported: a run that dies\n    after the cache is built has spent the expensive half and scores nothing, so the\n    guard around `main` swallows it and leaves the benchmark file behind. This one is\n    the exception. Training on the weaker labels does not look like a failure - it\n    completes, writes a plausible submission, and differs only in a log line - so it has\n    to stop the run rather than be absorbed by a guard designed for crashes.\n    \"\"\"\n\n\ndef find_label_table():\n    \"\"\"Locate a mounted table of pre-read report labels, if one is attached.\n\n    The lexicon turns a report into labels by matching morphology, and its failure\n    mode is silence: on a phrasing it does not carry it emits no opinion rather than a\n    wrong one. Silence is measurable without any ground truth - for each (report,\n    finding) pair, did anything match? - and that measurement says the misses are\n    concentrated in particular languages rather than spread evenly, on findings a knee\n    report almost always comments on.\n\n    Enumerating morphology for nine languages is the wrong instrument for that. Reading\n    the sentence is the right one, and a language model reads it. Against the annotated\n    studies the difference is large and one-sided, so when such a table is mounted it is\n    preferred; when it is not, the lexicon runs and the pipeline is unchanged. Both paths\n    produce the same columns, so nothing downstream knows which one supplied them.\n    \"\"\"\n    base = Path(\"/kaggle/input\")\n    cands = []\n    if base.is_dir():\n        for root, dirs, files in os.walk(base):\n            dirs[:] = [d for d in dirs if d not in (\"train_series\", \"test_series\")]\n            cands += [Path(root) / f for f in files if f.startswith(\"report_labels\")\n                      and f.endswith(\".csv\")]\n    cands += [p for p in (Path(\"data/derived/report_labels_v2.csv\"),) if p.is_file()]\n    for c in cands:\n        try:\n            head = pd.read_csv(c, nrows=1)\n        except Exception:\n            continue\n        if \"StudyInstanceUID\" in head.columns and all(t in head.columns for t in TARGETS):\n            return c\n    return None\n\n\ndef label_mount_attached():\n    \"\"\"True when an input directory was attached that is meant to carry a label table.\n\n    The fallback below is deliberate and has to stay silent for a run with no table\n    attached, because that is the ordinary case for anyone reading this notebook. It\n    must not stay silent for the other case: a table was attached and could not be used.\n    Those two are indistinguishable from the labels alone - both end with the lexicon -\n    so they are separated here by whether the mount exists at all.\n    \"\"\"\n    base = Path(\"/kaggle/input\")\n    if not base.is_dir():\n        return False\n    return any(\"label\" in p.name.lower() for p in base.iterdir() if p.is_dir())\n\n\ndef read_labels(train_df):\n    \"\"\"Labels for every training study, from a mounted table or from the lexicon.\n\n    Studies the mounted table does not cover fall back to the lexicon rather than being\n    dropped, so a partial table degrades coverage instead of losing rows.\n    \"\"\"\n    n = len(train_df)\n    lab = pd.DataFrame([extract(r) for r in train_df[\"Report\"].fillna(\"\")])\n    lab[\"StudyInstanceUID\"] = train_df[\"StudyInstanceUID\"].values\n    lab = lab.set_index(\"StudyInstanceUID\")\n\n    src = find_label_table()\n    if src is None:\n        if label_mount_attached():\n            raise LabelSourceError(\n                \"LABEL SOURCE: a label dataset is mounted but no usable table was found \"\n                \"in it. Falling back to the lexicon here would train on the weaker \"\n                \"labels and say so only in a log line, so the run stops instead.\")\n        log(f\"LABEL SOURCE: lexicon, {n} studies (no table mounted)\")\n        return lab\n\n    tab = pd.read_csv(src).set_index(\"StudyInstanceUID\")\n    missing = [c for c in LABEL_COLS if c not in tab.columns]\n    if missing:\n        raise LabelSourceError(\n            f\"LABEL SOURCE: {src} is missing {len(missing)} expected columns \"\n            f\"(first: {missing[0]!r}). Refusing to fall back silently.\")\n    hit = lab.index.intersection(tab.index)\n    if not len(hit):\n        raise LabelSourceError(\n            f\"LABEL SOURCE: {src} shares no StudyInstanceUID with train.csv.\")\n    log(f\"LABEL SOURCE: {src.name} covers {len(hit)} of {n} studies, \"\n        f\"lexicon for the remaining {n - len(hit)}\")\n    lab.loc[hit, LABEL_COLS] = tab.loc[hit, LABEL_COLS].values\n    return lab\n\n\nROOT = find_root()\nlog(f\"input root: {ROOT}\")\n\n\nIMG = CACHE_IMG           \n\n\ndef available_gb():\n    \"\"\"Memory this machine will actually lend, read rather than assumed.\n\n    A hardcoded ceiling is a guess about a machine the author is not sitting at, and a\n    guess that is too low costs coverage silently while a guess that is too high ends the\n    run. The machine will say, so it is asked.\n    \"\"\"\n    try:\n        with open(\"/proc/meminfo\") as fh:\n            info = {k.strip(): v for k, v in\n                    (l.split(\":\", 1) for l in fh if \":\" in l)}\n        return int(info[\"MemAvailable\"].split()[0]) / 1024 ** 2\n    except Exception:\n        return CACHE_BUDGET_GB / CACHE_FRACTION     \n\n\ndef plan_cache(n_study, n_test=0):\n    \"\"\"Choose how many slices per slot the memory the machine has will allow.\n\n    The cache is n_study x n_slot x slices x IMG^2 bytes. Coverage is the cheap axis -\n    linear - and resolution the expensive one, so when the budget binds it is the slice\n    count that gives way rather than the pixel grid. Deciding once, from the training\n    corpus size, keeps train and test caches on the same group layout.\n\n    Only a fraction of what is free is taken. The rest is not slack: the encoder, its\n    activations, the pinned batches and the frames all come out of the same pool, and the\n    cache is the one allocation big enough that overshooting it kills the run outright.\n    \"\"\"\n    avail = available_gb()\n    budget = min(avail * CACHE_FRACTION, CACHE_BUDGET_MAX_GB)\n\n    n_total = n_study + max(n_test, int(TEST_SHARE * n_study))\n    per_slice = n_total * N_SLOT * IMG * IMG\n    afford = int(budget * 1024 ** 3 // max(per_slice, 1))\n    groups = max(1, min(N_GROUP_MAX, afford // GROUP))\n    log(f\"memory: {avail:.1f} GB available, {budget:.1f} GB to the cache; \"\n        f\"sizing for {n_study} train + {n_total - n_study} test studies \"\n        f\"-> {groups} group(s) of {GROUP} = {groups * GROUP} slices per slot\"\n        + (f\" (wanted {N_GROUP_MAX})\" if groups < N_GROUP_MAX else \"\"))\n    return groups\n\n\nN_GROUP = plan_cache(len(pd.read_csv(ROOT / \"train.csv\")),\n                     len(pd.read_csv(ROOT / \"test.csv\")))\nCACHE_SLICES = GROUP * N_GROUP\nlog(f\"cache layout: {N_GROUP} groups x {GROUP} slices = {CACHE_SLICES} per slot\")\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:11.52099Z","iopub.status.busy":"2026-08-07T07:49:11.520439Z","iopub.status.idle":"2026-08-07T07:49:11.758526Z","shell.execute_reply":"2026-08-07T07:49:11.757451Z"},"papermill":{"duration":0.271878,"end_time":"2026-08-07T07:49:11.760385+00:00","exception":false,"start_time":"2026-08-07T07:49:11.488507+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-15","cell_type":"code","source":"HDR_TAGS = [\"SeriesDescription\", \"SequenceName\", \"ScanOptions\", \"ScanningSequence\",\n            \"RepetitionTime\", \"EchoTime\", \"Laterality\", \"PixelSpacing\", \"Rows\",\n            \"Columns\", \"RescaleSlope\", \"RescaleIntercept\",\n            \"ImagePositionPatient\", \"ImageOrientationPatient\"]\n\n\ndef _hdr_vec(s, n):\n    \"\"\"Parse a DICOM multi-value string as stored by probe(): floats joined by `|`.\"\"\"\n    if not isinstance(s, str):\n        return None\n    try:\n        v = [float(x) for x in s.split(\"|\")]\n    except ValueError:\n        return None\n    return np.array(v) if len(v) >= n else None\n\n\ndef side_from_geometry(h):\n    \"\"\"Study -> 'L' / 'R' / None, from where the image sits in the patient.\n\n    `Laterality` (0020,0060) is Type 2C and may legitimately be absent; in this corpus it\n    is missing on exactly half the studies, and the vendors it is missing from are whole\n    vendors rather than scattered series. A study with no tag is not a left knee, but the\n    normalisation upstream treats it as one, so half the corpus was never normalised and\n    the five side-defined targets - the two menisci, the two tibiofemoral compartments\n    and the medial collateral ligament - saw that axis reversed on a large minority of it.\n\n    The patient coordinate system fixes this without the tag: +x is the patient's left, so\n    the centre of a right knee sits at negative x. The centre is used rather than\n    `ImagePositionPatient` itself because that is the corner of the image, which is offset\n    by half a field of view - enough to change the sign on a knee near the midline.\n\n    The median over a study's series is what is thresholded, not a single series: probe()\n    reads one arbitrary slice per series, which on a sagittal stack can sit anywhere\n    across the joint. Studies whose centre falls near the midline are left unresolved\n    rather than guessed - measured against the tagged half, the rule is right 97% of the\n    time overall and no better than chance inside 20 mm.\n    \"\"\"\n    cx = {}\n    for r in h.itertuples(index=False):\n        ipp = _hdr_vec(getattr(r, \"ImagePositionPatient\", None), 3)\n        iop = _hdr_vec(getattr(r, \"ImageOrientationPatient\", None), 6)\n        ps = _hdr_vec(getattr(r, \"PixelSpacing\", None), 2)\n        rows, cols = getattr(r, \"Rows\", None), getattr(r, \"Columns\", None)\n        if ipp is None or iop is None or ps is None or not rows or not cols:\n            continue\n        try:\n            c = ipp[:3] + iop[:3] * ps[1] * float(cols) / 2 + iop[3:6] * ps[0] * float(rows) / 2\n        except (TypeError, ValueError):\n            continue\n        cx.setdefault(r.StudyInstanceUID, []).append(float(c[0]))\n    out = {}\n    for st, xs in cx.items():\n        m = float(np.median(xs))\n        out[st] = None if abs(m) < LAT_MIN_OFFSET_MM else (\"R\" if m < 0 else \"L\")\n    return out\n\n\ndef lat_of(h, tag=\"\"):\n    \"\"\"Study -> 'L' / 'R' / None: the tag where it exists, geometry where it does not.\n\n    The tag is present on exactly half the studies here and is sometimes an empty\n    string rather than absent, which is not the same as NaN. Treating the other half\n    as left-sided is what `normalise_laterality` did by omission, so the geometry\n    fallback is not a refinement - it is the difference between normalising half the\n    corpus and normalising all of it.\n    \"\"\"\n    geo = side_from_geometry(h)\n    d, n_tag, n_geo, n_none, n_disagree = {}, 0, 0, 0, 0\n    for st, g in h.groupby(\"StudyInstanceUID\"):\n        v = [str(x).strip().upper() for x in g[\"Laterality\"].dropna()]\n        v = [x[0] for x in v if x and x[0] in (\"L\", \"R\")]\n        side = v[0] if v else None\n        if side is not None:\n            n_tag += 1\n            if geo.get(st) is not None and geo[st] != side:\n                n_disagree += 1\n        else:\n            side = geo.get(st)\n            n_geo += side is not None\n            n_none += side is None\n        d[st] = side\n    log(f\"{tag}laterality: {n_tag} from the tag, {n_geo} from geometry, \"\n        f\"{n_none} unresolved; tag and geometry disagree on {n_disagree} \"\n        f\"({n_disagree / max(n_tag, 1):.1%} of the tagged)\")\n    return d\n\n\n\ndef probe(item):\n    split, study, series, path = item\n    row = {\"split\": split, \"StudyInstanceUID\": study, \"SeriesInstanceUID\": series,\n           \"dir\": path}\n    try:\n        files = sorted(e.name for e in os.scandir(path) if e.name.endswith(\".dcm\"))\n        row[\"files\"] = files\n        row[\"n_slices\"] = len(files)\n        if not files:\n            return row\n        ds = pydicom.dcmread(os.path.join(path, files[len(files) // 2]),\n                             stop_before_pixels=True, force=True)\n        for t in HDR_TAGS:\n            v = getattr(ds, t, None)\n            if v is None:\n                row[t] = None\n            elif isinstance(v, (list, tuple)) or type(v).__name__ == \"MultiValue\":\n                row[t] = \"|\".join(str(x) for x in v)\n            else:\n                row[t] = str(v)\n    except Exception as exc:\n        row[\"err\"] = str(exc)[:120]\n    return row\n\n\ndef walk(split):\n    \"\"\"Every series directory of a split, with one header read per series.\n\n    An absent split returns an empty frame *with the columns annotate expects*. Returning\n    a bare DataFrame looks like the same thing and is not: the next call indexes\n    `SeriesDescription` and raises KeyError, so the branch that exists to survive a\n    missing split is what turns it into a crash.\n    \"\"\"\n    base = ROOT / split\n    items = []\n    if not base.is_dir():\n        return pd.DataFrame(columns=[\"split\", \"StudyInstanceUID\", \"SeriesInstanceUID\",\n                                     \"dir\", \"files\", \"n_slices\"] + HDR_TAGS)\n    for study in os.scandir(base):\n        if study.is_dir():\n            for series in os.scandir(study.path):\n                if series.is_dir():\n                    items.append((split, study.name, series.name, series.path))\n    with ThreadPoolExecutor(max_workers=HDR_THREADS) as pool:\n        rows = list(pool.map(probe, items))\n    return pd.DataFrame(rows)\n\n\ndef annotate(df):\n    \"\"\"Recover fat suppression and pulse-sequence weighting from the header.\"\"\"\n    desc = (df[\"SeriesDescription\"].fillna(\"\") + \" \" + df[\"SequenceName\"].fillna(\"\"))\n    desc = desc.str.lower().str.replace(_SEP, \" \", regex=True)\n\n    opts = df[\"ScanOptions\"].fillna(\"\").str.upper().str.split(\"|\")\n\n    opts_fs = opts.apply(lambda ts: any(t.strip() in FATSAT_OPTS for t in ts))\n    df[\"fatsat\"] = desc.str.contains(_FATSAT_RX) | opts_fs\n\n    tr = pd.to_numeric(df[\"RepetitionTime\"], errors=\"coerce\")\n    te = pd.to_numeric(df[\"EchoTime\"], errors=\"coerce\")\n    gre = df[\"ScanningSequence\"].fillna(\"\").str.upper().str.contains(\"GR\")\n    t1, t2, pdw = desc.str.contains(_T1_RX), desc.str.contains(_T2_RX), desc.str.contains(_PD_RX)\n\n    df[\"weight\"] = np.where(t1 & ~t2 & ~pdw, \"T1\",\n                     np.where(t2 & ~pdw, \"T2\",\n                       np.where(pdw, \"PD\",\n                         np.where(gre, \"GRE\",\n                           np.where(tr < 800, \"T1\",\n                             np.where(te > 60, \"T2\",\n                               np.where(tr >= 800, \"PD\", \"UNK\")))))))\n    df[\"fluid\"] = np.isin(df[\"weight\"], [\"PD\", \"T2\"])\n    df[\"px\"] = pd.to_numeric(\n        df[\"PixelSpacing\"].fillna(\"\").str.split(\"|\").str[0].replace(\"\", np.nan),\n        errors=\"coerce\")\n    return df\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:11.825193Z","iopub.status.busy":"2026-08-07T07:49:11.824877Z","iopub.status.idle":"2026-08-07T07:49:11.847656Z","shell.execute_reply":"2026-08-07T07:49:11.846901Z"},"papermill":{"duration":0.05747,"end_time":"2026-08-07T07:49:11.849402+00:00","exception":false,"start_time":"2026-08-07T07:49:11.791932+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-16","cell_type":"code","source":"def pick_slots(series_df, plane_map):\n    \"\"\"One series per slot per study.\n\n    Ties are broken toward the stack with the most slices: a thicker stack samples the\n    joint more densely, and the three-slice sampler below benefits from the margin.\n    \"\"\"\n    series_df = series_df.copy()\n    series_df[\"plane\"] = series_df[\"SeriesInstanceUID\"].map(plane_map)\n    out = {}\n    for study, g in series_df.groupby(\"StudyInstanceUID\"):\n        chosen = {}\n        for name, plane, fluid, fs in SLOTS:\n            sel = (g[\"plane\"] == plane) & (g[\"fatsat\"] == fs)\n\n            if fluid is not None:\n                sel &= (g[\"fluid\"] == fluid)\n            cand = g[sel]\n\n            if len(cand):\n                chosen[name] = cand.sort_values(\"n_slices\", ascending=False).iloc[0]\n        out[study] = chosen\n    return out\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:11.913925Z","iopub.status.busy":"2026-08-07T07:49:11.913067Z","iopub.status.idle":"2026-08-07T07:49:11.919981Z","shell.execute_reply":"2026-08-07T07:49:11.919252Z"},"papermill":{"duration":0.041162,"end_time":"2026-08-07T07:49:11.921566+00:00","exception":false,"start_time":"2026-08-07T07:49:11.880404+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-19","cell_type":"code","source":"ORDER_TAGS = [(0x0020, 0x0032), (0x0020, 0x0037), (0x0020, 0x0013)]\n\nDECODE_FAILED = []\n\n\ndef order_slices(rec):\n    \"\"\"Return the series' files sorted along the through-plane axis.\n\n    A DICOM file name here is a SOP Instance UID, which is assigned arbitrarily. Sorting\n    by it therefore produces an order uncorrelated with anatomy - measured over one\n    series, Spearman between file-name rank and physical position is 0.009, i.e. none.\n    Anything that assumes the file order means something is then operating on noise: the\n    three channels of a \"2.5D\" input are three unrelated views rather than neighbouring\n    slices, \"the middle of the stack\" is a random subset, and reversing slice order to\n    normalise laterality reverses nothing meaningful.\n\n    The physical order is recoverable exactly. Each slice carries its position in patient\n    coordinates and the in-plane axes; projecting the position onto the slice normal\n    gives a signed through-plane coordinate, monotonic along the stack:\n\n        n = r_x  x  r_y ,      k = p . n\n\n    `InstanceNumber` is the fallback. It usually tracks the projection up to sign, but\n    interleaved and multi-echo acquisitions need not number slices in the order they\n    occupy in space - but the projection is signed in patient\n    coordinates, which is what laterality normalisation needs.\n    \"\"\"\n    files, d = rec[\"files\"], rec[\"dir\"]\n    keyed = []\n    for f in files:\n        k = None\n        try:\n            ds = pydicom.dcmread(os.path.join(d, f), force=True, stop_before_pixels=True,\n                                 specific_tags=ORDER_TAGS)\n            iop = np.asarray(ds.ImageOrientationPatient, dtype=float)\n            ipp = np.asarray(ds.ImagePositionPatient, dtype=float)\n            k = float(np.dot(ipp, np.cross(iop[:3], iop[3:])))\n        except Exception:\n            try:\n                k = float(ds.InstanceNumber)\n            except Exception:\n                k = None\n        keyed.append((k, f))\n    if any(k is None for k, _ in keyed):\n\n        return files, False\n    return [f for _, f in sorted(keyed, key=lambda t: t[0])], True\n\n\ndef read_slot(rec, n_slice=None, out_size=None):\n    \"\"\"`n_slice` physically spread slices from one series, at `out_size` pixels.\n\n    Returns uint8 [n_slice, out, out] normalised per-series to its 1st-99th\n    percentile. Percentiles rather than min/max because MR intensity has no absolute\n    scale and a single bright vessel would otherwise compress the whole dynamic range.\n\n    Reading is the expensive half of this pipeline, so the caller reads once at the\n    largest configuration it needs and derives the smaller ones from the returned buffer\n    rather than re-reading.\n    \"\"\"\n    n_slice = GROUP if n_slice is None else n_slice\n    out_size = IMG if out_size is None else out_size\n    files, d, px = rec.get(\"ordered\") or rec[\"files\"], rec[\"dir\"], rec[\"px\"]\n    n = len(files)\n    if n == 0:\n        return None\n\n    lo, hi = int(SLICE_BAND[0] * (n - 1)), int(SLICE_BAND[1] * (n - 1))\n    idx = np.unique(np.linspace(lo, hi, n_slice).astype(int)) if hi > lo else np.array([n // 2])\n    while len(idx) < n_slice:\n        idx = np.append(idx, idx[-1])\n\n    planes = []\n    for i in idx[:n_slice]:\n        try:\n            ds = pydicom.dcmread(os.path.join(d, files[int(i)]), force=True)\n            a = ds.pixel_array.astype(np.float32)\n            sl = float(getattr(ds, \"RescaleSlope\", 1) or 1)\n            ic = float(getattr(ds, \"RescaleIntercept\", 0) or 0)\n            a = a * sl + ic\n        except Exception:\n            a = None                      # no shape is known here; see below\n        planes.append(a)\n\n    got = [k for k, p in enumerate(planes) if p is not None]\n    if not got:\n        DECODE_FAILED.append(rec.get(\"SeriesInstanceUID\", d))\n        return None\n    if len(got) < len(planes):\n        DECODE_FAILED.append(rec.get(\"SeriesInstanceUID\", d))\n        for k, p in enumerate(planes):\n            if p is None:\n                planes[k] = planes[min(got, key=lambda j: abs(j - k))]\n\n    shp = planes[0].shape\n    planes = [p if p.shape == shp else np.zeros(shp, np.float32) for p in planes]\n    vol = np.stack(planes)\n\n    if px and np.isfinite(px) and px > 0:\n        want = int(round(CROP_MM / px))\n        h, w = shp\n        if 16 < want < min(h, w):\n            cy, cx = h // 2, w // 2\n            half = want // 2\n            vol = vol[:, max(0, cy - half):cy + half, max(0, cx - half):cx + half]\n\n    lo_v, hi_v = np.percentile(vol, [1, 99])\n    vol = np.clip((vol - lo_v) / max(hi_v - lo_v, 1e-6), 0, 1)\n\n    t = torch.from_numpy(np.ascontiguousarray(vol)).unsqueeze(0)\n    t = F.interpolate(t, size=(out_size, out_size), mode=\"bilinear\", align_corners=False)\n\n    return (t.squeeze(0) * 255).round().clamp(0, 255).to(torch.uint8)\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:12.105366Z","iopub.status.busy":"2026-08-07T07:49:12.105092Z","iopub.status.idle":"2026-08-07T07:49:12.122912Z","shell.execute_reply":"2026-08-07T07:49:12.122301Z"},"papermill":{"duration":0.050046,"end_time":"2026-08-07T07:49:12.124321+00:00","exception":false,"start_time":"2026-08-07T07:49:12.074275+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-21","cell_type":"code","source":"def normalise_laterality(img, plane, lat):\n    \"\"\"Map every knee onto a left-knee convention.\n\n    Coronal and axial views mirror under a horizontal flip. Sagittal stacks are not\n    mirror images of each other - the slice order runs medial-to-lateral in opposite\n    directions - so the channel order is reversed instead.\n    \"\"\"\n    if lat != \"R\":\n        return img\n    if plane in (\"Coronal\", \"Axial\"):\n        return torch.flip(img, dims=[-1])\n    return torch.flip(img, dims=[0])\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:12.246523Z","iopub.status.busy":"2026-08-07T07:49:12.24624Z","iopub.status.idle":"2026-08-07T07:49:12.250976Z","shell.execute_reply":"2026-08-07T07:49:12.250139Z"},"papermill":{"duration":0.037233,"end_time":"2026-08-07T07:49:12.252479+00:00","exception":false,"start_time":"2026-08-07T07:49:12.215246+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-23","cell_type":"code","source":"def build_cache(slot_map, plane_map, lat_map, tag):\n    \"\"\"Decode every (study, slot) once into an in-memory uint8 array.\n\n    Fine-tuning revisits the same pixels every epoch. Reading them from the mount each\n    time would make the epoch count a function of I/O rather than of learning, so they\n    are decoded once and held as bytes: intensity has already been normalised into\n    [0, 1], and eight bits cost nothing a bilinear resize has not already cost.\n\n    CACHE_SLICES positions are kept per slot, which the training loop reads as N_GROUP\n    groups of GROUP consecutive channels.\n    \"\"\"\n    studies = sorted(slot_map)\n    sidx = {s: i for i, s in enumerate(studies)}\n    cache = np.zeros((len(studies), N_SLOT, CACHE_SLICES, IMG, IMG), np.uint8)\n    mask = np.zeros((len(studies), N_SLOT), np.float32)\n    log(f\"{tag}: cache {cache.shape} = {cache.nbytes / 1024 ** 3:.1f} GB\")\n\n    jobs = [(st, k, plane, slot_map[st][name])\n            for st in studies\n            for k, (name, plane, _, _) in enumerate(SLOTS)\n            if name in slot_map[st]]\n\n\n    t_ord = time.time()\n    n_slice_total = sum(len(j[3][\"files\"]) for j in jobs)\n    log(f\"{tag}: ordering {len(jobs)} slot-series ({n_slice_total} slice headers)\")\n    ok = done = 0\n    CHUNK_O = 1024\n    with ThreadPoolExecutor(max_workers=ORDER_THREADS) as pool:\n        for c0 in range(0, len(jobs), CHUNK_O):\n            block = jobs[c0:c0 + CHUNK_O]\n            for (_, _, _, rec), (files, good) in zip(\n                    block, pool.map(lambda j: order_slices(j[3]), block)):\n                rec[\"ordered\"] = files\n                ok += int(good)\n                done += 1\n\n            budget = min(ORDER_BUDGET_S, max(60.0, (TIME_BUDGET - (time.time() - T0)) * 0.35))\n            if time.time() - t_ord > budget:\n                log(f\"{tag}: ordering budget spent at {done}/{len(jobs)}; \"\n                    f\"the rest keep file order\")\n                break\n    log(f\"{tag}: ordered {ok}/{len(jobs)} by geometry \"\n        f\"({len(jobs) - ok} kept arbitrary) in {time.time() - t_ord:.0f}s\")\n\n    log(f\"{tag}: decoding {len(jobs)} slot-series\")\n    n_failed_before = len(DECODE_FAILED)\n\n    CHUNK = 512\n    done = 0\n    with ThreadPoolExecutor(max_workers=PIX_THREADS) as pool:\n        for c0 in range(0, len(jobs), CHUNK):\n            block = jobs[c0:c0 + CHUNK]\n            for (st, k, plane, _), img in zip(\n                    block, pool.map(lambda j: read_slot(j[3], CACHE_SLICES, IMG), block)):\n                done += 1\n                if img is None:\n                    continue\n                cache[sidx[st], k] = normalise_laterality(img, plane,\n                                                          lat_map.get(st)).numpy()\n                mask[sidx[st], k] = 1.0\n            if done % 4096 < CHUNK:\n                log(f\"  {tag} {done}/{len(jobs)}\")\n            if time.time() - T0 > TIME_BUDGET:\n                log(f\"  {tag}: time budget reached during decode\")\n                break\n    n_failed = len(DECODE_FAILED) - n_failed_before\n    log(f\"{tag}: {int(mask.sum())}/{len(jobs)} slots filled\"\n        + (f\"; {n_failed} series had a slice that would not decode\" if n_failed else \"\"))\n    gc.collect()\n    return studies, cache, mask\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:12.372012Z","iopub.status.busy":"2026-08-07T07:49:12.371395Z","iopub.status.idle":"2026-08-07T07:49:12.383556Z","shell.execute_reply":"2026-08-07T07:49:12.382876Z"},"papermill":{"duration":0.044933,"end_time":"2026-08-07T07:49:12.385188+00:00","exception":false,"start_time":"2026-08-07T07:49:12.340255+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-25","cell_type":"code","source":"class SlotHead(nn.Module):\n    \"\"\"Per-diagnosis attention over the slot embeddings of one study.\n\n    Each finding is read on particular sequences - cruciates sagittally, collateral\n    ligaments and the meniscal body coronally, patellar cartilage axially - so pooling\n    the slots identically would dilute the one that carries the evidence with the rest.\n\n    The aggregation is deliberately this simple. With a study-level label there is no\n    signal telling the model which part of a study matters, so extra attention\n    parameters below the slot level would have nothing to learn from and would spend\n    their capacity fitting noise.\n    \"\"\"\n\n    def __init__(self, dim, n_slot, n_out, hidden=256, p=0.2):\n        super().__init__()\n        self.proj = nn.Sequential(nn.LayerNorm(dim), nn.Linear(dim, hidden), nn.GELU())\n        self.slot_emb = nn.Parameter(torch.randn(n_slot, hidden) * 0.02)\n        self.query = nn.Parameter(torch.randn(n_out, hidden) * 0.02)\n        self.drop = nn.Dropout(p)\n        self.out = nn.Linear(hidden, n_out)\n        self.hidden = hidden\n\n    def forward(self, x, mask):\n        h = self.proj(x) + self.slot_emb\n        att = torch.einsum(\"bsh,oh->bos\", h, self.query) / self.hidden ** 0.5\n        att = att.masked_fill(mask.unsqueeze(1) < 0.5, -1e4).softmax(-1)\n        ctx = self.drop(torch.einsum(\"bos,bsh->boh\", att, h))\n        return (ctx * self.out.weight.unsqueeze(0)).sum(-1) + self.out.bias\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:12.507074Z","iopub.status.busy":"2026-08-07T07:49:12.506663Z","iopub.status.idle":"2026-08-07T07:49:12.513676Z","shell.execute_reply":"2026-08-07T07:49:12.512876Z"},"papermill":{"duration":0.040155,"end_time":"2026-08-07T07:49:12.515406+00:00","exception":false,"start_time":"2026-08-07T07:49:12.475251+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-26","cell_type":"code","source":"class Model(nn.Module):\n    \"\"\"Encoder plus head, trained end to end.\n\n    A study arrives as a bag of slot images. The bag is flattened for the encoder and\n    folded back before the head, so the encoder never sees the study structure and the\n    head never sees pixels.\n    \"\"\"\n\n    def __init__(self, backbone, dim):\n        super().__init__()\n        self.backbone = backbone\n        self.head = SlotHead(dim, N_SLOT, len(TARGETS))\n        self.register_buffer(\"mean\", torch.tensor([0.485, 0.456, 0.406]).view(1, 3, 1, 1))\n        self.register_buffer(\"std\", torch.tensor([0.229, 0.224, 0.225]).view(1, 3, 1, 1))\n\n    def forward(self, imgs, mask, img_size=None):\n        B, S = imgs.shape[:2]\n        x = imgs.reshape(B * S, *imgs.shape[2:]).float().div_(255.0)\n        if img_size is not None and img_size != x.shape[-1]:\n\n            x = F.interpolate(x, size=(img_size, img_size), mode=\"bilinear\",\n                              align_corners=False)\n        x = (x - self.mean) / self.std\n        out = self.backbone(pixel_values=x).last_hidden_state\n        feat = torch.cat([out[:, 0], out[:, 1:].mean(1)], dim=1).reshape(B, S, -1)\n        return self.head(feat, mask)\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:12.580997Z","iopub.status.busy":"2026-08-07T07:49:12.580439Z","iopub.status.idle":"2026-08-07T07:49:12.588763Z","shell.execute_reply":"2026-08-07T07:49:12.587896Z"},"papermill":{"duration":0.044918,"end_time":"2026-08-07T07:49:12.590631+00:00","exception":false,"start_time":"2026-08-07T07:49:12.545713+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-28","cell_type":"code","source":"def build_model(unfreeze_last, source=None, variant=\"small\"):\n    \"\"\"Load the encoder and open the last `unfreeze_last` blocks for training.\n\n    The early blocks of a self-supervised transformer are generic edge and texture\n    filters; the late blocks carry semantics. Opening only the late ones is the cautious\n    choice - there may not be enough supervision here to improve the early ones and there\n    is certainly enough to damage them - but how far the line should sit is a question\n    the corpus has to answer rather than the intuition.\n\n    `source` names where the weights come from. Left unset it is the attached model\n    directory, which is the only thing available here. It is a parameter so that a run\n    off the platform builds the same object from the same code rather than from a second\n    definition that has to be kept in step by hand.\n    \"\"\"\n    from transformers import AutoModel\n    p = source if source is not None else find_dinov2(variant)\n    if p is None:\n        raise FileNotFoundError(\"DINOv2 weights not attached\")\n    bb = AutoModel.from_pretrained(str(p))\n    n_layer = len(bb.encoder.layer)\n    for prm in bb.parameters():\n        prm.requires_grad = False\n    for blk in bb.encoder.layer[max(0, n_layer - unfreeze_last):]:\n        for prm in blk.parameters():\n            prm.requires_grad = True\n    for prm in bb.layernorm.parameters():\n        prm.requires_grad = True\n    dim = bb.config.hidden_size * 2\n    trainable = sum(p.numel() for p in bb.parameters() if p.requires_grad)\n    log(f\"backbone: {n_layer} blocks, last {unfreeze_last} trainable \"\n        f\"({trainable / 1e6:.1f}M params), feature dim {dim}\")\n    return Model(bb, dim)\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:12.726744Z","iopub.status.busy":"2026-08-07T07:49:12.726374Z","iopub.status.idle":"2026-08-07T07:49:12.73349Z","shell.execute_reply":"2026-08-07T07:49:12.732782Z"},"papermill":{"duration":0.042709,"end_time":"2026-08-07T07:49:12.735225+00:00","exception":false,"start_time":"2026-08-07T07:49:12.692516+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-29","cell_type":"code","source":"def take_group(cache_rows, g):\n    \"\"\"Slice GROUP consecutive channels out of the cached slices.\"\"\"\n    return cache_rows[:, :, g * GROUP:(g + 1) * GROUP]\n\n\ndef augment(imgs):\n    \"\"\"A small rigid jitter and an intensity scale, applied to a whole bag at once.\n\n    Neither flip is available here, and for different reasons. A horizontal flip would\n    reintroduce the nuisance axis that the laterality normalisation removed - it would\n    undo, once per batch, what the header pass was run to establish.\n\n    A vertical flip is not a nuisance axis at all. A knee is acquired in a canonical\n    orientation, and no study in this corpus looks like its own vertical mirror. An\n    augmentation is meant to cover directions along which the label does not change; this\n    one moves the input off the distribution the encoder will be asked about, which is a\n    different thing. Where a finding sits in the frame is also information rather than\n    noise - a Baker cyst is identified by lying in the popliteal fossa, not by its\n    appearance alone.\n\n    What is left is jitter that no label depends on: a few degrees of rotation, a few\n    per cent of scale and translation. That still prevents memorising the exact framing,\n    which is what an augmentation is for, while leaving the anatomy where it was.\n    \"\"\"\n\n    lead = imgs.shape[:-3]\n    x = imgs.reshape(-1, *imgs.shape[-3:]).float()\n    n, dev = x.shape[0], x.device\n\n    rot = (torch.rand(n, device=dev) - 0.5) * 2 * (AUG_ROT_DEG * np.pi / 180)\n\n    sc = 1.0 + torch.rand(n, device=dev) * AUG_SCALE\n    tx = (torch.rand(n, device=dev) - 0.5) * 2 * AUG_SHIFT\n    ty = (torch.rand(n, device=dev) - 0.5) * 2 * AUG_SHIFT\n    cos, sin = torch.cos(rot) / sc, torch.sin(rot) / sc\n    theta = torch.zeros(n, 2, 3, device=dev, dtype=torch.float32)\n    theta[:, 0, 0], theta[:, 0, 1], theta[:, 0, 2] = cos, -sin, tx\n    theta[:, 1, 0], theta[:, 1, 1], theta[:, 1, 2] = sin, cos, ty\n    grid = F.affine_grid(theta, x.shape, align_corners=False)\n    x = F.grid_sample(x, grid, mode=\"bilinear\", padding_mode=\"border\", align_corners=False)\n\n    scale = 1.0 + (torch.rand(n, 1, 1, 1, device=dev) - 0.5) * 2 * AUG_INTENSITY\n    x = (x * scale).clamp(0, 255)\n    return x.reshape(*lead, *x.shape[-3:]).to(imgs.dtype)\n\n\n@torch.no_grad()\ndef predict(model, cache, mask, idx, dev, img_size=None):\n    \"\"\"Average the logits over the groups of each slot.\n\n    Training sees one group at a time, which acts as augmentation along the stack;\n    inference averages over all of them, so the prediction does not depend on which\n    group a single draw happened to pick. Where the cache holds one group per slot the\n    two coincide.\n    \"\"\"\n    model.eval()\n    out = []\n    for b in range(0, len(idx), EVAL_BATCH):\n        sel = idx[b:b + EVAL_BATCH]\n        rows = torch.from_numpy(cache[sel]).to(dev)\n        m = torch.from_numpy(mask[sel]).to(dev)\n        acc = None\n        for g in range(N_GROUP):\n            with torch.autocast(\"cuda\", enabled=dev.type == \"cuda\"):\n                z = model(take_group(rows, g), m, img_size).float()\n            acc = z if acc is None else acc + z\n        out.append(torch.sigmoid(acc / N_GROUP).cpu().numpy())\n    return np.concatenate(out) if out else np.zeros((0, len(TARGETS)), np.float32)\n\n\ndef macro_auc(y, p):\n    from sklearn.metrics import roc_auc_score\n    return float(np.nanmean([roc_auc_score(y[:, j], p[:, j])\n                             if len(set(y[:, j])) > 1 else np.nan\n                             for j in range(y.shape[1])]))\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:12.799926Z","iopub.status.busy":"2026-08-07T07:49:12.799156Z","iopub.status.idle":"2026-08-07T07:49:12.81253Z","shell.execute_reply":"2026-08-07T07:49:12.811877Z"},"papermill":{"duration":0.047757,"end_time":"2026-08-07T07:49:12.814074+00:00","exception":false,"start_time":"2026-08-07T07:49:12.766317+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-31","cell_type":"code","source":"def write_submission(pred, studies, test_df, path):\n    \"\"\"Write one submission file from a prediction matrix.\n\n    Predictions are converted to per-column ranks first: the metric reads only order, so\n    ranks discard nothing, and they make files from different configurations directly\n    comparable and safe to average.\n    \"\"\"\n    sub = pd.DataFrame(pd.DataFrame(pred).rank(pct=True).values, columns=TARGETS)\n    sub.insert(0, \"StudyInstanceUID\", studies)\n    sub = test_df[[\"StudyInstanceUID\"]].merge(sub, on=\"StudyInstanceUID\", how=\"left\")\n    sub[TARGETS] = sub[TARGETS].fillna(0.5)\n    sub.to_csv(path, index=False)\n    return sub\n\n\ndef write_benchmark_submission():\n    \"\"\"Write the 0.5 benchmark file immediately.\n\n    A submission that never writes scores nothing at all, which is strictly worse than\n    scoring badly. The try/except around main() covers exceptions, but a kill for memory\n    is a SIGKILL and never reaches it. So a valid file exists from the first second and\n    is overwritten only once real predictions are ready.\n    \"\"\"\n    t = pd.read_csv(ROOT / \"test.csv\")\n    for c in TARGETS:\n        t[c] = 0.5\n    t.to_csv(\"submission.csv\", index=False)\n\n\ndef main():\n    write_benchmark_submission()\n\n\n    read_labels(pd.read_csv(ROOT / \"train.csv\", usecols=[\"StudyInstanceUID\", \"Report\"]))\n\n    test_df = pd.read_csv(ROOT / \"test.csv\")\n    test_series = pd.read_csv(ROOT / \"test_series.csv\")\n    train_df = pd.read_csv(ROOT / \"train.csv\")\n    train_series = pd.read_csv(ROOT / \"train_series.csv\")\n    log(f\"train {train_df.shape} test {test_df.shape}\")\n\n    both = pd.concat([train_series, test_series])\n    plane_map = dict(zip(both[\"SeriesInstanceUID\"], both[\"Anatomical_Plane\"]))\n\n    log(\"header pass: test\")\n    hte = annotate(walk(\"test_series\"))\n    log(f\"  {len(hte)} test series\")\n    log(\"header pass: train\")\n    htr = annotate(walk(\"train_series\"))\n    log(f\"  {len(htr)} train series\")\n\n    slots_te, slots_tr = pick_slots(hte, plane_map), pick_slots(htr, plane_map)\n    cov = pd.Series([len(v) for v in slots_tr.values()]).describe()\n    log(f\"train slots per study: mean {cov['mean']:.2f} min {cov['min']:.0f} \"\n        f\"max {cov['max']:.0f}\")\n\n    st_tr, Ctr, Mtr = build_cache(slots_tr, plane_map, lat_of(htr, \"train \"), \"train\")\n    st_te, Cte, Mte = build_cache(slots_te, plane_map, lat_of(hte, \"test \"), \"test\")\n\n    t_lab = time.time()\n    lab = read_labels(train_df)\n    log(f\"derived labels for {len(lab)} studies in {time.time() - t_lab:.1f}s\")\n\n    gold = train_df.set_index(\"StudyInstanceUID\")[TARGETS]\n    gold = gold[gold.notna().all(axis=1)]\n\n    Y = np.zeros((len(st_tr), len(TARGETS)), np.float32)\n    W = np.zeros_like(Y)\n    for i, st in enumerate(st_tr):\n        if st in gold.index:\n            Y[i], W[i] = gold.loc[st].values, 3.0\n        elif st in lab.index:\n            r = lab.loc[st]\n            Y[i] = r[TARGETS].values\n            W[i] = 0.25 + 0.75 * r[[t + \"__conf\" for t in TARGETS]].values\n    keep = np.where(W.sum(1) > 0)[0]\n    log(f\"supervised {len(keep)} of {len(st_tr)} studies (annotated {len(gold)})\")\n\n\n    import hashlib\n    rep = train_df.set_index(\"StudyInstanceUID\")[\"Report\"].fillna(\"\")\n    grp = np.array([int(hashlib.md5(rep.get(s, s).encode()).hexdigest()[:8], 16) % 5\n                    for s in st_tr])\n    va = np.array([i for i in keep if grp[i] == 0])\n    tr = np.array([i for i in keep if grp[i] != 0])\n    if len(va) == 0 or len(tr) < BATCH_STUDIES:\n        cut = max(1, len(keep) // 5)\n        va, tr = keep[:cut], keep[cut:]\n    log(f\"train {len(tr)} / holdout {len(va)} studies\")\n\n    gpos = {s: i for i, s in enumerate(st_tr)}\n    va_set = set(va.tolist())\n    gi = np.array([gpos[s] for s in gold.index if s in gpos and gpos[s] in va_set])\n    gold_y = gold.loc[[st_tr[i] for i in gi]].values.astype(int) if len(gi) else None\n    yv = (Y[va] > 0.5).astype(int)\n    log(f\"annotation check: {len(gi)} of {len(gold)} annotated studies are in the holdout\")\n\n    dev = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    results, test_preds = {}, {}\n\n    for cfg in RUNS:\n        pitch = CROP_MM / cfg[\"img\"]\n        log(f\"=== {cfg['name']}: {cfg['img']} px, {pitch:.3f} mm/pixel, \"\n            f\"{pitch * 14:.2f} mm per patch token ===\")\n        torch.manual_seed(SEED)\n        model = build_model(UNFREEZE_LAST).to(dev)\n        opt = torch.optim.AdamW([\n            {\"params\": [p for p in model.backbone.parameters() if p.requires_grad],\n             \"lr\": LR_BACKBONE},\n            {\"params\": model.head.parameters(), \"lr\": LR_HEAD},\n        ], weight_decay=WEIGHT_DECAY)\n        steps = max(EPOCHS * (len(tr) // BATCH_STUDIES), 1)\n        sched = torch.optim.lr_scheduler.OneCycleLR(\n            opt, max_lr=[LR_BACKBONE, LR_HEAD], total_steps=steps, pct_start=0.15)\n        scaler = torch.amp.GradScaler(\"cuda\", enabled=dev.type == \"cuda\")\n\n        best, best_state, best_annot = -1.0, None, float(\"nan\")\n        for ep in range(EPOCHS):\n            model.train()\n            perm = np.random.permutation(tr)\n            tot, nstep = 0.0, 0\n            for b in range(0, len(perm) - BATCH_STUDIES + 1, BATCH_STUDIES):\n                sel = perm[b:b + BATCH_STUDIES]\n                rows = torch.from_numpy(Ctr[sel]).to(dev)\n                g = int(torch.randint(N_GROUP, (1,)).item())\n                imgs = augment(take_group(rows, g))\n                m = torch.from_numpy(Mtr[sel]).to(dev)\n                y = torch.from_numpy(Y[sel]).to(dev)\n                w = torch.from_numpy(W[sel]).to(dev)\n                with torch.autocast(\"cuda\", enabled=dev.type == \"cuda\"):\n                    loss = (F.binary_cross_entropy_with_logits(\n                        model(imgs, m, cfg[\"img\"]), y, reduction=\"none\") * w).mean()\n                opt.zero_grad(set_to_none=True)\n                scaler.scale(loss).backward()\n                scaler.step(opt)\n                scaler.update()\n                sched.step()\n                tot += loss.item()\n                nstep += 1\n\n            pv = predict(model, Ctr, Mtr, va, dev, cfg[\"img\"])\n            d = macro_auc(yv, pv)\n            g_auc = float(\"nan\")\n            if gold_y is not None and len(gi):\n                g_auc = macro_auc(gold_y, predict(model, Ctr, Mtr, gi, dev, cfg[\"img\"]))\n            log(f\"  epoch {ep + 1}/{EPOCHS}  loss {tot / max(nstep, 1):.4f}\"\n                f\"  holdout {d:.4f}  annot(n={len(gi)}) {g_auc:.4f}\")\n\n\n            if d > best:\n                best, best_annot = d, g_auc\n                best_state = {k: v.detach().cpu().clone()\n                              for k, v in model.state_dict().items()}\n            if time.time() - T0 > TIME_BUDGET:\n                log(\"  time budget reached\")\n                break\n\n        if best_state is not None:\n            model.load_state_dict(best_state)\n        results[cfg[\"name\"]] = (best, best_annot)\n        test_preds[cfg[\"name\"]] = predict(model, Cte, Mte, np.arange(len(st_te)), dev,\n                                          cfg[\"img\"])\n        log(f\"  {cfg['name']}: best holdout {best:.4f} (annot {best_annot:.4f})\")\n        del model, opt, sched, scaler, best_state\n        gc.collect()\n        if dev.type == \"cuda\":\n            torch.cuda.empty_cache()\n\n    log(\"---- summary ----\")\n    for n, (d, g_auc) in results.items():\n        log(f\"  {n:12s} holdout {d:.4f}   annot {g_auc:.4f}\")\n    pick = max(results, key=lambda k: results[k][0])\n    log(f\"best on the holdout: {pick} ({results[pick][0]:.4f})\")\n\n    for name, pred in test_preds.items():\n        sub = write_submission(pred, st_te, test_df, f\"submission_{name}.csv\")\n        log(f\"  submission_{name}.csv {sub.shape}; \"\n            f\"nulls {int(sub[TARGETS].isna().sum().sum())}\")\n\n    ens = np.mean([pd.DataFrame(p).rank(pct=True).values for p in test_preds.values()],\n                  axis=0)\n    write_submission(ens, st_te, test_df, \"submission_rankmean.csv\")\n    log(f\"  submission_rankmean.csv (rank mean of {len(test_preds)})\")\n\n    sub = write_submission(test_preds[pick], st_te, test_df, \"submission.csv\")\n    log(f\"submission.csv = {pick}; {sub.shape}; \"\n        f\"nulls {int(sub[TARGETS].isna().sum().sum())}\")\n    print(sub.head().to_string())\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:12.942903Z","iopub.status.busy":"2026-08-07T07:49:12.942565Z","iopub.status.idle":"2026-08-07T07:49:12.971077Z","shell.execute_reply":"2026-08-07T07:49:12.97035Z"},"papermill":{"duration":0.063583,"end_time":"2026-08-07T07:49:12.972674+00:00","exception":false,"start_time":"2026-08-07T07:49:12.909091+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"code-32","cell_type":"code","source":"try:\n    main()\nexcept LabelSourceError:\n    traceback.print_exc()\n    raise\nexcept Exception:\n    traceback.print_exc()\n\n    t = pd.read_csv(find_root() / \"test.csv\")\n    for c in TARGETS:\n        t[c] = 0.5\n    t.to_csv(\"submission.csv\", index=False)\n    print(\"wrote fallback submission.csv\")\nlog(\"done\")\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-08-07T07:49:13.038477Z","iopub.status.busy":"2026-08-07T07:49:13.037417Z","iopub.status.idle":"2026-08-07T09:09:41.018126Z","shell.execute_reply":"2026-08-07T09:09:41.017168Z"},"papermill":{"duration":4828.016433,"end_time":"2026-08-07T09:09:41.019989+00:00","exception":false,"start_time":"2026-08-07T07:49:13.003556+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}