{"cells":[{"cell_type":"code","execution_count":null,"metadata":{},"outputs":[],"source":"import os\nimport sys\nimport glob\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nTARGETS = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\",\n    \"Medial OA\", \"Lateral OA\", \"PF OA\", \"Effusion\",\n    \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\"\n]\n\ndef find_root():\n    for p in [\n        Path(\"/kaggle/input/competitions/rsna-knee-abnormality-detection\"),\n        Path(\"/kaggle/input/rsna-knee-abnormality-detection\"),\n    ]:\n        if p.exists() and (p / \"test.csv\").exists():\n            return p\n    for root, dirs, files in os.walk(\"/kaggle/input\"):\n        if \"test.csv\" in files:\n            return Path(root)\n    raise FileNotFoundError(\"Could not locate RSNA Knee dataset root.\")\n\nROOT = find_root()\nprint(f\"Located competition root: {ROOT}\")\n\ntest_df = pd.read_csv(ROOT / \"test.csv\")\nprint(f\"Found {len(test_df)} test studies.\")\n\n# Clinical regex rule lexicon for MRI knee radiology reports\nFINDING_KEYWORDS = {\n    \"ACL\": [\"anterior cruciate\", \"acl tear\", \"acl disruption\", \"acl rupture\", \"acl sprain\", \"acl graft\"],\n    \"MCL\": [\"medial collateral\", \"mcl tear\", \"mcl sprain\", \"mcl disruption\", \"mcl injury\"],\n    \"Medial Meniscus\": [\"medial meniscus\", \"medial meniscal\", \"mm tear\", \"posterior horn medial meniscus\"],\n    \"Lateral Meniscus\": [\"lateral meniscus\", \"lateral meniscal\", \"lm tear\", \"posterior horn lateral meniscus\"],\n    \"Medial OA\": [\"medial joint space narrowing\", \"medial compartment osteoarthritis\", \"medial compartment degenerative\", \"medial compartment oa\"],\n    \"Lateral OA\": [\"lateral joint space narrowing\", \"lateral compartment osteoarthritis\", \"lateral compartment degenerative\", \"lateral compartment oa\"],\n    \"PF OA\": [\"patellofemoral osteoarthritis\", \"patellofemoral degenerative\", \"patellofemoral oa\", \"chondromalacia patellae\", \"trochlear groove narrowing\"],\n    \"Effusion\": [\"joint effusion\", \"joint fluid\", \"suprapatellar effusion\", \"moderate effusion\", \"large effusion\"],\n    \"Synovitis\": [\"synovitis\", \"synovial thickening\", \"synovial proliferation\", \"synovial hypertrophy\", \"synovial enhancement\"],\n    \"Baker's\": [\"baker cyst\", \"baker's cyst\", \"popliteal cyst\", \"gastrocnemius-semimembranosus bursa\"],\n    \"Contusion\": [\"bone contusion\", \"bone marrow edema\", \"bone bruise\", \"trabecular microfracture\"],\n    \"Fracture\": [\"fracture\", \"cortical break\", \"subchondral fracture\", \"tibial plateau fracture\", \"femoral fracture\", \"patellar fracture\"]\n}\n\nNEGATIONS = [\"no \", \"without \", \"intact\", \"normal\", \"unremarkable\", \"free of\", \"negative for\", \"rules out\", \"no evidence\"]\n\ndef extract_report_scores(report_text):\n    scores = {}\n    if not isinstance(report_text, str) or len(report_text.strip()) == 0:\n        return {t: 0.5 for t in TARGETS}\n    \n    text = report_text.lower()\n    for target in TARGETS:\n        keywords = FINDING_KEYWORDS.get(target, [])\n        score = 0.5\n        pos_hits = sum(1 for kw in keywords if kw in text)\n        \n        # Check if keyword is negated\n        neg_hits = 0\n        for kw in keywords:\n            if kw in text:\n                pos = text.find(kw)\n                preceding = text[max(0, pos - 40):pos]\n                if any(neg in preceding for neg in NEGATIONS):\n                    neg_hits += 1\n        \n        if pos_hits > 0 and neg_hits == 0:\n            score = 0.90 + 0.03 * min(pos_hits, 3)\n        elif neg_hits > 0:\n            score = 0.10 - 0.02 * min(neg_hits, 3)\n        scores[target] = score\n    return scores\n\n# Multi-plane DICOM feature extraction\ntest_series_dir = ROOT / \"test_series\"\n\nresults = []\nfor idx, row in test_df.iterrows():\n    study_id = str(row[\"StudyInstanceUID\"])\n    report = row.get(\"Report\", \"\")\n    \n    # 1. Report features\n    report_probs = extract_report_scores(report)\n    \n    # 2. DICOM volume intensity analysis\n    study_dir = test_series_dir / study_id\n    dicom_mod = {t: 0.0 for t in TARGETS}\n    \n    if study_dir.exists():\n        dcm_files = sorted(list(study_dir.glob(\"*/*.dcm\")) + list(study_dir.glob(\"*.dcm\")))\n        if len(dcm_files) > 0:\n            step = max(1, len(dcm_files) // 8)\n            sampled = dcm_files[::step][:8]\n            vals = []\n            for sf in sampled:\n                try:\n                    ds = pydicom.dcmread(str(sf), stop_before_pixels=False)\n                    vals.append(np.mean(ds.pixel_array))\n                except Exception:\n                    pass\n            if len(vals) > 0:\n                v_std = float(np.std(vals))\n                v_mean = float(np.mean(vals))\n                # High contrast slices correspond to effusion/synovitis fluid signals\n                dicom_mod[\"Effusion\"] += (v_std / 500.0) * 0.10\n                dicom_mod[\"Synovitis\"] += (v_std / 500.0) * 0.08\n                dicom_mod[\"Medial Meniscus\"] += (v_mean / 2000.0) * 0.05\n    \n    row_pred = {}\n    for t in TARGETS:\n        base_p = report_probs[t]\n        h_jitter = (int(study_id.replace(\".\", \"\")[-4:]) % 1000) / 50000.0\n        final_p = np.clip(base_p + dicom_mod[t] + h_jitter, 0.001, 0.999)\n        row_pred[t] = final_p\n        \n    row_pred[\"StudyInstanceUID\"] = study_id\n    results.append(row_pred)\n\nsub_df = pd.DataFrame(results)\n\n# Apply dense continuous percentile ranking across test studies\nranked_df = pd.DataFrame()\nranked_df[\"StudyInstanceUID\"] = sub_df[\"StudyInstanceUID\"]\nfor t in TARGETS:\n    ranked_df[t] = sub_df[t].rank(pct=True, method=\"dense\").clip(0.01, 0.99)\n\n# Align perfectly with test.csv\nfinal_sub = test_df[[\"StudyInstanceUID\"]].merge(ranked_df, on=\"StudyInstanceUID\", how=\"left\")\nfinal_sub[TARGETS] = final_sub[TARGETS].fillna(0.5)\n\nfinal_sub.to_csv(\"submission.csv\", index=False)\nprint(\"Successfully generated submission.csv with multi-target ranking!\")\nprint(\"Shape:\", final_sub.shape)\nprint(final_sub.head())\n"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.12"}},"nbformat":4,"nbformat_minor":4}