{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport re\nfrom tqdm.auto import tqdm\n\n# Configuration\nKAGGLE_INPUT_DIR = \"/kaggle/input\"\nTRAIN_CSV = os.path.join(KAGGLE_INPUT_DIR, 'rsna-knee-abnormality-detection', 'train.csv')\ntry:\n    for root, dirs, files in os.walk('/kaggle/input'):\n        if 'train.csv' in files:\n            TRAIN_CSV = os.path.join(root, 'train.csv')\n            break\nexcept Exception:\n    pass\nOUTPUT_CSV = 'train_pseudo_labeled.csv'\n\nTARGET_COLS = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\",\n    \"Medial OA\", \"Lateral OA\", \"PF OA\", \"Effusion\",\n    \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\"\n]\n\n# Robust finding patterns\nFINDING_TERMS = {\n    \"ACL\": [\"acl tear\", \"anterior cruciate ligament tear\", \"sprain of acl\", \"torn acl\", \"high grade acl\"],\n    \"MCL\": [\"mcl tear\", \"medial collateral ligament tear\", \"sprain of mcl\", \"torn mcl\", \"high grade mcl\"],\n    \"Medial Meniscus\": [\"medial meniscus tear\", \"tear of the medial meniscus\", \"medial meniscal tear\", \"macerated medial meniscus\"],\n    \"Lateral Meniscus\": [\"lateral meniscus tear\", \"tear of the lateral meniscus\", \"lateral meniscal tear\", \"macerated lateral meniscus\"],\n    \"Medial OA\": [\"medial compartment osteoarthritis\", \"medial joint space narrowing\", \"medial compartment cartilage loss\", \"medial oa\"],\n    \"Lateral OA\": [\"lateral compartment osteoarthritis\", \"lateral joint space narrowing\", \"lateral compartment cartilage loss\", \"lateral oa\"],\n    \"PF OA\": [\"patellofemoral osteoarthritis\", \"patellofemoral joint space narrowing\", \"patellofemoral cartilage loss\", \"pf oa\"],\n    \"Effusion\": [\"joint effusion\", \"knee effusion\", \"moderate effusion\", \"large effusion\"],\n    \"Synovitis\": [\"synovitis\", \"synovial thickening\", \"synovial proliferation\"],\n    \"Baker's\": [\"baker's cyst\", \"baker cyst\", \"popliteal cyst\"],\n    \"Contusion\": [\"bone contusion\", \"bone bruise\", \"marrow edema\", \"contusion\"],\n    \"Fracture\": [\"fracture\", \"fx\", \"cortical step-off\", \"avulsion\"]\n}\n\n# Anatomical aliases for checking negative mentions\nTARGET_ALIASES = {\n    \"ACL\": [\"acl\", \"anterior cruciate ligament\"],\n    \"MCL\": [\"mcl\", \"medial collateral ligament\"],\n    \"Medial Meniscus\": [\"medial meniscus\", \"medial meniscal\"],\n    \"Lateral Meniscus\": [\"lateral meniscus\", \"lateral meniscal\"],\n    \"Medial OA\": [\"medial compartment\", \"medial joint space\"],\n    \"Lateral OA\": [\"lateral compartment\", \"lateral joint space\"],\n    \"PF OA\": [\"patellofemoral\", \"pf compartment\"],\n    \"Effusion\": [\"effusion\", \"joint fluid\"],\n    \"Synovitis\": [\"synovium\", \"synovial\"],\n    \"Baker's\": [\"popliteal fossa\", \"baker's cyst\", \"baker cyst\"],\n    \"Contusion\": [\"marrow\", \"bone contusion\"],\n    \"Fracture\": [\"fracture\", \"fx\"]\n}\n\ndef parse_report_robust(text, patterns, aliases):\n    if not isinstance(text, str) or not text.strip():\n        return np.nan\n    text = text.lower()\n    \n    # 1. Search for positive finding phrases\n    for pat in patterns:\n        for match in re.finditer(rf'\\b{pat}\\b', text):\n            start, end = match.span()\n            # Context window: 35 chars before and 25 chars after\n            window = text[max(0, start - 35):min(len(text), end + 25)]\n            if re.search(r'\\b(no|not|without|free of|negative for|intact|unremarkable|normal)\\b', window):\n                return 0.1 # Negated finding\n            return 0.9 # True positive\n            \n    # 2. Check if structure is explicitly documented as normal\n    for alias in aliases:\n        for match in re.finditer(rf'\\b{alias}\\b', text):\n            start, end = match.span()\n            window = text[max(0, start - 35):min(len(text), end + 25)]\n            if re.search(r'\\b(intact|normal|unremarkable|preserved|visualized without)\\b', window):\n                return 0.1 # True negative\n                \n    return np.nan\n\ndef generate_pseudo_labels():\n    print(\"Loading original dataset...\")\n    try:\n        df = pd.read_csv(TRAIN_CSV)\n    except FileNotFoundError:\n        print(f\"Warning: Could not find {TRAIN_CSV}. Make sure you are running on Kaggle.\")\n        # Create a dummy for testing\n        df = pd.DataFrame({\"StudyInstanceUID\": [\"dummy_1\"], \"Report\": [\"Intact ACL. Medial meniscus tear present.\"], **{k: [np.nan] for k in TARGET_COLS}})\n    \n    print(\"Analyzing reports for missing labels...\")\n    \n    if 'loss_weight' not in df.columns:\n        df['loss_weight'] = 1.0 \n    \n    annotated_count = 0\n    pseudo_count = 0\n    \n    for idx, row in tqdm(df.iterrows(), total=len(df)):\n        # If any of the labels are missing, attempt to parse the report\n        if pd.isna(row[TARGET_COLS]).any():\n            if pd.notna(row['Report']):\n                found_any = False\n                for target in TARGET_COLS:\n                    if pd.isna(row[target]):\n                        pred = parse_report_robust(row['Report'], FINDING_TERMS[target], TARGET_ALIASES[target])\n                        if pd.notna(pred):\n                            df.at[idx, target] = pred\n                            found_any = True\n                \n                if found_any:\n                    # Only downweight if it previously had NO human annotations\n                    if pd.isna(row[TARGET_COLS]).all():\n                        df.at[idx, 'loss_weight'] = 0.5 \n                    pseudo_count += 1\n        else:\n            annotated_count += 1\n\n    print(f\"\\nSummary:\")\n    print(f\"Original Fully Human Annotated Studies: {annotated_count}\")\n    print(f\"Studies updated with NLP Pseudo-Labels: {pseudo_count}\")\n    print(f\"Total Usable Studies: {annotated_count + pseudo_count}\")\n    \n    df.to_csv(OUTPUT_CSV, index=False)\n    print(f\"Saved to {OUTPUT_CSV}! The training script will now automatically detect and use this dataset.\")\n\nif __name__ == \"__main__\":\n    generate_pseudo_labels()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T15:53:05.406945Z","iopub.execute_input":"2026-08-20T15:53:05.407211Z","iopub.status.idle":"2026-08-20T15:53:14.515511Z","shell.execute_reply.started":"2026-08-20T15:53:05.407189Z","shell.execute_reply":"2026-08-20T15:53:14.514687Z"}},"outputs":[],"execution_count":null}]}