{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nBASE = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\n\nprint(\"INPUT DATA CHECK\")\nprint(\"Exists:\", os.path.exists(BASE))\n\nfor name in os.listdir(BASE):\n    path = os.path.join(BASE, name)\n    print(name, \"->\", \"DIR\" if os.path.isdir(path) else \"FILE\")\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kaggle competitions submit -c rsna-knee-abnormality-detection -f submission.csv -k roshnihusj/<NOTEBOOK> -v <VERSION> -m \"Message\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-09T08:55:44.679988Z","iopub.execute_input":"2026-09-09T08:55:44.680593Z","iopub.status.idle":"2026-09-09T08:55:44.687019Z","shell.execute_reply.started":"2026-09-09T08:55:44.680555Z","shell.execute_reply":"2026-09-09T08:55:44.685646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# RSNA KNEE — ONE-CELL REPORT WEAK-LABEL PIPELINE\n# Safe/read-only on /kaggle/input\n# Creates checkpoint in /kaggle/working\n# ============================================================\n\nimport os, re, json, pickle, warnings\nimport numpy as np\nimport pandas as pd\nfrom sklearn.metrics import roc_auc_score\n\nwarnings.filterwarnings(\"ignore\")\n\n# -----------------------------\n# 1. Locate competition data\n# -----------------------------\nBASE = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\nTRAIN = os.path.join(BASE, \"train.csv\")\n\nif not os.path.exists(TRAIN):\n    raise FileNotFoundError(f\"train.csv not found at {TRAIN}\")\n\ntrain = pd.read_csv(TRAIN)\n\nLABEL_COLS = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\",\n    \"Medial OA\", \"Lateral OA\", \"PF OA\", \"Effusion\",\n    \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\"\n]\n\nif \"Report\" not in train.columns:\n    raise RuntimeError(\"The train.csv file does not contain a Report column.\")\n\n# -----------------------------\n# 2. Gold-labelled subset\n# -----------------------------\ngold = train[train[LABEL_COLS].notna().all(axis=1)].copy()\n\nprint(\"=\" * 65)\nprint(\"RSNA KNEE — REPORT WEAK-LABEL CHECK\")\nprint(\"=\" * 65)\nprint(f\"Full training studies : {len(train):,}\")\nprint(f\"Gold-labelled studies : {len(gold):,}\")\nprint(f\"Reports available     : {train['Report'].notna().sum():,}\")\nprint()\n\nif len(gold) == 0:\n    raise RuntimeError(\"No fully labelled studies found.\")\n\n# -----------------------------\n# 3. Normalize reports\n# -----------------------------\ndef clean_report(x):\n    if pd.isna(x):\n        return \"\"\n    x = str(x).lower()\n    x = x.replace(\"\\n\", \" \")\n    x = re.sub(r\"\\s+\", \" \", x)\n    return x.strip()\n\ntrain[\"_report_clean\"] = train[\"Report\"].apply(clean_report)\n\n# -----------------------------\n# 4. Conservative medical\n#    phrase scoring\n# -----------------------------\n#\n# Score is deliberately continuous:\n#   0 = strong negative evidence\n#   1 = strong positive evidence\n#\n# This is NOT yet our final weak-label model.\n# It is a validation checkpoint.\n# -----------------------------\n\npatterns = {\n\n    \"ACL\": {\n        \"pos\": [\n            r\"\\bacl\\b.{0,60}(tear|ruptur|injur|sprain|thick|abnormal)\",\n            r\"(tear|ruptur|injur|sprain|thick|abnormal).{0,60}\\bacl\\b\",\n            r\"anterior cruciate ligament.{0,60}(tear|ruptur|injur|abnormal)\"\n        ],\n        \"neg\": [\n            r\"\\bacl\\b.{0,40}(intact|normal|preserved)\",\n            r\"(intact|normal|preserved).{0,40}\\bacl\\b\"\n        ]\n    },\n\n    \"MCL\": {\n        \"pos\": [\n            r\"\\bmcl\\b.{0,60}(tear|ruptur|injur|sprain|thick|abnormal)\",\n            r\"(tear|ruptur|injur|sprain|thick|abnormal).{0,60}\\bmcl\\b\",\n            r\"medial collateral ligament.{0,60}(tear|ruptur|injur|abnormal)\"\n        ],\n        \"neg\": [\n            r\"\\bmcl\\b.{0,40}(intact|normal|preserved)\",\n            r\"(intact|normal|preserved).{0,40}\\bmcl\\b\"\n        ]\n    },\n\n    \"Medial Meniscus\": {\n        \"pos\": [\n            r\"medial meniscus.{0,80}(tear|tear\\s*extending|ruptur|extrusion|abnormal|degener)\",\n            r\"(tear|ruptur|extrusion|abnormal).{0,80}medial meniscus\"\n        ],\n        \"neg\": [\n            r\"medial meniscus.{0,50}(intact|normal|preserved|unremarkable)\"\n        ]\n    },\n\n    \"Lateral Meniscus\": {\n        \"pos\": [\n            r\"lateral meniscus.{0,80}(tear|tear\\s*extending|ruptur|extrusion|abnormal|degener)\",\n            r\"(tear|ruptur|extrusion|abnormal).{0,80}lateral meniscus\"\n        ],\n        \"neg\": [\n            r\"lateral meniscus.{0,50}(intact|normal|preserved|unremarkable)\"\n        ]\n    },\n\n    \"Medial OA\": {\n        \"pos\": [\n            r\"medial compartment.{0,100}(osteoarthritis|arthrosis|cartilage loss|cartilage defect|joint space narrowing)\",\n            r\"medial.{0,80}(osteoarthritis|arthrosis|cartilage loss|joint space narrowing)\",\n            r\"(osteoarthritis|arthrosis).{0,80}medial\"\n        ],\n        \"neg\": []\n    },\n\n    \"Lateral OA\": {\n        \"pos\": [\n            r\"lateral compartment.{0,100}(osteoarthritis|arthrosis|cartilage loss|cartilage defect|joint space narrowing)\",\n            r\"lateral.{0,80}(osteoarthritis|arthrosis|cartilage loss|joint space narrowing)\",\n            r\"(osteoarthritis|arthrosis).{0,80}lateral\"\n        ],\n        \"neg\": []\n    },\n\n    \"PF OA\": {\n        \"pos\": [\n            r\"(patellofemoral|patello[- ]femoral).{0,100}(osteoarthritis|arthrosis|cartilage loss|cartilage defect|joint space narrowing)\",\n            r\"(osteoarthritis|arthrosis).{0,100}(patellofemoral|patello[- ]femoral)\",\n            r\"patellar.{0,80}(cartilage loss|cartilage defect|chondral|arthrosis)\",\n            r\"trochlear.{0,80}(cartilage loss|cartilage defect|chondral|arthrosis)\"\n        ],\n        \"neg\": []\n    },\n\n    \"Effusion\": {\n        \"pos\": [\n            r\"(joint|knee).{0,50}(effusion|fluid)\",\n            r\"(effusion|fluid).{0,50}(joint|knee)\"\n        ],\n        \"neg\": [\n            r\"(no|without|minimal).{0,20}(joint )?effusion\",\n            r\"no significant effusion\"\n        ]\n    },\n\n    \"Synovitis\": {\n        \"pos\": [\n            r\"synovitis\",\n            r\"synovial.{0,50}(thickening|proliferation)\",\n            r\"(synovial|synovium).{0,50}(enhancement|abnormal)\"\n        ],\n        \"neg\": [\n            r\"no synovitis\",\n            r\"without synovitis\"\n        ]\n    },\n\n    \"Baker's\": {\n        \"pos\": [\n            r\"baker.{0,40}(cyst|collection)\",\n            r\"popliteal cyst\",\n            r\"popliteal.{0,50}(cyst|fluid)\"\n        ],\n        \"neg\": [\n            r\"no baker\",\n            r\"no popliteal cyst\",\n            r\"without popliteal cyst\"\n        ]\n    },\n\n    \"Contusion\": {\n        \"pos\": [\n            r\"bone contusion\",\n            r\"bone bruise\",\n            r\"osseous contusion\",\n            r\"marrow.{0,40}(edema|contusion|bruise)\"\n        ],\n        \"neg\": []\n    },\n\n    \"Fracture\": {\n        \"pos\": [\n            r\"\\bfracture\\b\",\n            r\"fractured\",\n            r\"fracture line\",\n            r\"cortical.{0,40}(break|disruption)\"\n        ],\n        \"neg\": [\n            r\"no fracture\",\n            r\"without fracture\",\n            r\"fracture.*not identified\"\n        ]\n    }\n}\n\n# -----------------------------\n# 5. Negation-aware scoring\n# -----------------------------\ndef score_label(text, spec):\n\n    if not text:\n        return 0.5\n\n    pos_hits = 0\n    neg_hits = 0\n\n    for p in spec[\"pos\"]:\n        try:\n            pos_hits += len(re.findall(p, text))\n        except Exception:\n            pass\n\n    for p in spec[\"neg\"]:\n        try:\n            neg_hits += len(re.findall(p, text))\n        except Exception:\n            pass\n\n    # Strong negative evidence\n    if neg_hits > 0 and pos_hits == 0:\n        return 0.08\n\n    # Strong positive evidence\n    if pos_hits > 0 and neg_hits == 0:\n        return min(0.55 + 0.12 * pos_hits, 0.98)\n\n    # Conflicting evidence: stay uncertain\n    if pos_hits > 0 and neg_hits > 0:\n        return 0.50 + 0.05 * min(pos_hits - neg_hits, 2)\n\n    # No evidence\n    return 0.45\n\n# -----------------------------\n# 6. Generate report scores\n# -----------------------------\nreport_scores = pd.DataFrame(index=train.index)\n\nfor label in LABEL_COLS:\n    report_scores[label] = [\n        score_label(text, patterns[label])\n        for text in train[\"_report_clean\"]\n    ]\n\n# -----------------------------\n# 7. Validate against gold labels\n# -----------------------------\nresults = []\n\nfor label in LABEL_COLS:\n    mask = gold.index\n\n    y_true = train.loc[mask, label].astype(float).values\n    y_score = report_scores.loc[mask, label].astype(float).values\n\n    # AUC requires both classes\n    if len(np.unique(y_true)) < 2:\n        auc = np.nan\n    else:\n        auc = roc_auc_score(y_true, y_score)\n\n    results.append({\n        \"Label\": label,\n        \"Gold_Positive\": int(y_true.sum()),\n        \"Report_AUC\": auc\n    })\n\nresults_df = pd.DataFrame(results)\n\nmacro_auc = results_df[\"Report_AUC\"].mean()\n\nprint(\"REPORT → GOLD VALIDATION\")\nprint(\"-\" * 65)\nprint(results_df.to_string(index=False))\nprint(\"-\" * 65)\nprint(f\"MACRO AUC: {macro_auc:.4f}\")\nprint()\n\n# -----------------------------\n# 8. Coverage statistics\n# -----------------------------\ncoverage = []\n\nfor label in LABEL_COLS:\n    s = report_scores[label]\n\n    coverage.append({\n        \"Label\": label,\n        \"Strong_Positive\": int((s >= 0.67).sum()),\n        \"Strong_Negative\": int((s <= 0.20).sum()),\n        \"Uncertain\": int(((s > 0.20) & (s < 0.67)).sum())\n    })\n\ncoverage_df = pd.DataFrame(coverage)\n\nprint(\"WEAK-LABEL COVERAGE — ALL TRAINING STUDIES\")\nprint(\"-\" * 65)\nprint(coverage_df.to_string(index=False))\nprint()\n\n# -----------------------------\n# 9. Save a SAFE checkpoint\n# -----------------------------\ncheckpoint = {\n    \"report_scores\": report_scores,\n    \"gold_indices\": gold.index.tolist(),\n    \"label_cols\": LABEL_COLS,\n    \"results\": results_df,\n    \"coverage\": coverage_df,\n    \"macro_auc\": float(macro_auc),\n}\n\nOUT = \"/kaggle/working/report_weak_labels_checkpoint.pkl\"\n\nwith open(OUT, \"wb\") as f:\n    pickle.dump(checkpoint, f)\n\n# Also save CSV for easy inspection\nCSV_OUT = \"/kaggle/working/report_weak_labels.csv\"\nreport_scores.to_csv(CSV_OUT, index=False)\n\nprint(\"=\" * 65)\nprint(\"CHECKPOINT SAVED\")\nprint(\"=\" * 65)\nprint(OUT)\nprint(CSV_OUT)\nprint()\nprint(f\"Macro AUC against 58 gold studies: {macro_auc:.4f}\")\n\nif macro_auc >= 0.80:\n    print(\"🔥 STRONG SIGNAL — proceed to the next training stage.\")\nelif macro_auc >= 0.70:\n    print(\"🟢 USEFUL SIGNAL — worth incorporating into training.\")\nelif macro_auc >= 0.60:\n    print(\"🟡 MODERATE SIGNAL — improve the report-labeling method first.\")\nelse:\n    print(\"🔴 WEAK SIGNAL — do NOT train on these weak labels yet.\")\n\nprint(\"=\" * 65)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-09T07:53:24.012295Z","iopub.execute_input":"2026-09-09T07:53:24.012631Z","iopub.status.idle":"2026-09-09T07:53:29.587Z","shell.execute_reply.started":"2026-09-09T07:53:24.0126Z","shell.execute_reply":"2026-09-09T07:53:29.586193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CLEAN FINAL RECOVERY — ONE CELL\n# ============================================================\n\nimport os\nimport pickle\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport pydicom\nimport torch\nimport torchvision.models as models\n\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import LeaveOneOut\nfrom sklearn.metrics import roc_auc_score\n\n# ------------------------------------------------------------\n# PATHS\n# ------------------------------------------------------------\n\nBASE = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\nWORK = \"/kaggle/working\"\n\nCHECKPOINT = os.path.join(\n    WORK, \"FINAL_RECOVERY_CHECKPOINT.pkl\"\n)\n\n# ------------------------------------------------------------\n# LABELS\n# ------------------------------------------------------------\n\nLABEL_COLS = [\n    \"ACL\",\n    \"MCL\",\n    \"Medial Meniscus\",\n    \"Lateral Meniscus\",\n    \"Medial OA\",\n    \"Lateral OA\",\n    \"PF OA\",\n    \"Effusion\",\n    \"Synovitis\",\n    \"Baker's\",\n    \"Contusion\",\n    \"Fracture\"\n]\n\n# ------------------------------------------------------------\n# LOAD CSVs\n# ------------------------------------------------------------\n\ntrain = pd.read_csv(\n    os.path.join(BASE, \"train.csv\")\n)\n\ntrain_series = pd.read_csv(\n    os.path.join(BASE, \"train_series.csv\")\n)\n\nprint(\"=\" * 60)\nprint(\"CLEAN FINAL RECOVERY\")\nprint(\"=\" * 60)\n\nprint(\"Train:\", train.shape)\nprint(\"Train series:\", train_series.shape)\n\n# ------------------------------------------------------------\n# GET THE 58 LABELED STUDIES\n# ------------------------------------------------------------\n\nlabeled = train[\n    train[LABEL_COLS].notna().all(axis=1)\n].copy()\n\n# We previously established that the labeled subset is 58.\nstudy_ids = labeled[\"StudyInstanceUID\"].tolist()\n\nY = labeled[\n    LABEL_COLS\n].values.astype(np.float32)\n\nprint(\"Labeled studies:\", len(study_ids))\nprint(\"Y shape:\", Y.shape)\n\nassert len(study_ids) == 58\n\n# ------------------------------------------------------------\n# MRI FUNCTIONS\n# ------------------------------------------------------------\n\ndef normalize_mri(img):\n\n    img = img.astype(np.float32)\n\n    lo = np.percentile(img, 1)\n    hi = np.percentile(img, 99)\n\n    if hi <= lo:\n        return np.zeros_like(\n            img,\n            dtype=np.float32\n        )\n\n    img = np.clip(\n        img,\n        lo,\n        hi\n    )\n\n    img = (\n        (img - lo) /\n        (hi - lo)\n    )\n\n    return img.astype(np.float32)\n\n\ndef resize_mri(img, size=224):\n\n    return cv2.resize(\n        img,\n        (size, size),\n        interpolation=cv2.INTER_AREA\n    ).astype(np.float32)\n\n\ndef load_series_images(\n    series_path,\n    n_slices=5\n):\n\n    records = []\n\n    for fname in os.listdir(series_path):\n\n        fpath = os.path.join(\n            series_path,\n            fname\n        )\n\n        if not os.path.isfile(fpath):\n            continue\n\n        try:\n\n            ds = pydicom.dcmread(\n                fpath\n            )\n\n            pixel = ds.pixel_array.astype(\n                np.float32\n            )\n\n            instance = getattr(\n                ds,\n                \"InstanceNumber\",\n                None\n            )\n\n            if instance is None:\n                instance = len(records)\n\n            records.append(\n                (int(instance), pixel)\n            )\n\n        except Exception:\n            continue\n\n    if len(records) == 0:\n        return None\n\n    records.sort(\n        key=lambda x: x[0]\n    )\n\n    indices = np.linspace(\n        0,\n        len(records) - 1,\n        n_slices,\n        dtype=int\n    )\n\n    selected = []\n\n    for idx in indices:\n\n        img = records[idx][1]\n\n        img = normalize_mri(img)\n\n        img = resize_mri(\n            img,\n            224\n        )\n\n        selected.append(img)\n\n    return np.stack(\n        selected\n    ).astype(np.float32)\n\n# ------------------------------------------------------------\n# RECONSTRUCT EXACTLY 5 SERIES PER STUDY\n# ------------------------------------------------------------\n\nprint(\"\\nReconstructing MRI data...\")\n\nall_mri_data = {}\n\nfor i, sid in enumerate(study_ids):\n\n    study_path = os.path.join(\n        BASE,\n        \"train_series\",\n        sid\n    )\n\n    series_ids = [\n        x for x in os.listdir(study_path)\n        if os.path.isdir(\n            os.path.join(\n                study_path,\n                x\n            )\n        )\n    ]\n\n    # Same deterministic ordering\n    series_ids = sorted(series_ids)\n\n    usable = []\n\n    for series_id in series_ids:\n\n        path = os.path.join(\n            study_path,\n            series_id\n        )\n\n        arr = load_series_images(\n            path,\n            n_slices=5\n        )\n\n        if arr is not None:\n            usable.append(arr)\n\n    if len(usable) < 5:\n\n        raise RuntimeError(\n            f\"{sid}: only \"\n            f\"{len(usable)} usable series\"\n        )\n\n    # Keep exactly 5\n    all_mri_data[sid] = usable[:5]\n\n    if (i + 1) % 10 == 0:\n        print(\n            f\"Processed {i + 1}/58 studies\"\n        )\n\nprint(\"\\nMRI reconstruction complete.\")\n\n# ------------------------------------------------------------\n# RESNET18\n# ------------------------------------------------------------\n\nprint(\"\\nLoading ResNet18...\")\n\nresnet = models.resnet18(\n    weights=models.ResNet18_Weights.DEFAULT\n)\n\nresnet.fc = torch.nn.Identity()\n\nresnet.eval()\n\n# ImageNet normalization\nIM_MEAN = torch.tensor(\n    [0.485, 0.456, 0.406],\n    dtype=torch.float32\n).view(1, 3, 1, 1)\n\nIM_STD = torch.tensor(\n    [0.229, 0.224, 0.225],\n    dtype=torch.float32\n).view(1, 3, 1, 1)\n\n# ------------------------------------------------------------\n# EXTRACT SERIES EMBEDDINGS\n# ------------------------------------------------------------\n\nprint(\"\\nExtracting embeddings...\")\n\nX_mean_list = []\nX_max_list = []\n\nfor i, sid in enumerate(study_ids):\n\n    series_embeddings = []\n\n    for series in all_mri_data[sid]:\n\n        images = torch.from_numpy(\n            series\n        ).float()\n\n        # (5,224,224)\n        # -> (5,3,224,224)\n\n        images = (\n            images\n            .unsqueeze(1)\n            .repeat(1, 3, 1, 1)\n        )\n\n        images = (\n            images - IM_MEAN\n        ) / IM_STD\n\n        with torch.no_grad():\n\n            embeddings = resnet(\n                images\n            )\n\n        # Average 5 slices\n        series_embedding = (\n            embeddings\n            .mean(dim=0)\n            .cpu()\n            .numpy()\n        )\n\n        series_embeddings.append(\n            series_embedding\n        )\n\n    series_embeddings = np.stack(\n        series_embeddings\n    )\n\n    # Mean across 5 series\n    X_mean_list.append(\n        series_embeddings.mean(\n            axis=0\n        )\n    )\n\n    # Max across 5 series\n    X_max_list.append(\n        series_embeddings.max(\n            axis=0\n        )\n    )\n\n    if (i + 1) % 10 == 0:\n        print(\n            f\"Embedded {i + 1}/58 studies\"\n        )\n\nX_mean = np.stack(\n    X_mean_list\n).astype(np.float32)\n\nX_max = np.stack(\n    X_max_list\n).astype(np.float32)\n\n# ------------------------------------------------------------\n# KNOWN CHAMPION\n# ------------------------------------------------------------\n\nX_weighted = (\n    0.7 * X_mean +\n    0.3 * X_max\n).astype(np.float32)\n\nprint(\"\\nRepresentations:\")\nprint(\"X_mean:\", X_mean.shape)\nprint(\"X_max:\", X_max.shape)\nprint(\n    \"X_weighted:\",\n    X_weighted.shape\n)\n\n# ------------------------------------------------------------\n# SAVE BEFORE EVALUATION\n# ------------------------------------------------------------\n\ncheckpoint = {\n    \"X_mean\": X_mean,\n    \"X_max\": X_max,\n    \"X_weighted\": X_weighted,\n    \"Y\": Y,\n    \"study_ids\": study_ids,\n    \"label_cols\": LABEL_COLS,\n    \"mean_weight\": 0.7,\n    \"max_weight\": 0.3\n}\n\nwith open(\n    CHECKPOINT,\n    \"wb\"\n) as f:\n\n    pickle.dump(\n        checkpoint,\n        f,\n        protocol=pickle.HIGHEST_PROTOCOL\n    )\n\nprint(\"\\n🛡️ CHECKPOINT SAVED\")\nprint(\n    \"File:\",\n    CHECKPOINT\n)\n\nprint(\n    \"Size MB:\",\n    round(\n        os.path.getsize(\n            CHECKPOINT\n        ) / 1024**2,\n        2\n    )\n)\n\n# ------------------------------------------------------------\n# EVALUATE CHAMPION\n# ------------------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 60)\nprint(\"CHAMPION EVALUATION\")\nprint(\"=\" * 60)\n\npredictions = np.zeros_like(Y)\n\nfor j, label in enumerate(\n    LABEL_COLS\n):\n\n    y = Y[:, j]\n\n    pred = np.zeros(\n        len(y),\n        dtype=np.float32\n    )\n\n    for tr, te in LeaveOneOut().split(\n        X_weighted\n    ):\n\n        model = make_pipeline(\n            StandardScaler(),\n            LogisticRegression(\n                C=0.01,\n                max_iter=5000,\n                class_weight=\"balanced\"\n            )\n        )\n\n        model.fit(\n            X_weighted[tr],\n            y[tr]\n        )\n\n        pred[te] = (\n            model\n            .predict_proba(\n                X_weighted[te]\n            )[0, 1]\n        )\n\n    predictions[:, j] = pred\n\n    auc = roc_auc_score(\n        y,\n        pred\n    )\n\n    print(\n        f\"{label:18s} {auc:.4f}\"\n    )\n\nmacro_auc = float(\n    np.mean([\n        roc_auc_score(\n            Y[:, j],\n            predictions[:, j]\n        )\n        for j in range(\n            Y.shape[1]\n        )\n    ])\n)\n\nprint(\"\\n\" + \"=\" * 60)\nprint(\n    f\"MACRO AUC: {macro_auc:.4f}\"\n)\nprint(\"=\" * 60)\n\n# ------------------------------------------------------------\n# SAVE FINAL EVALUATION\n# ------------------------------------------------------------\n\ncheckpoint[\n    \"predictions\"\n] = predictions\n\ncheckpoint[\n    \"macro_auc\"\n] = macro_auc\n\nwith open(\n    CHECKPOINT,\n    \"wb\"\n) as f:\n\n    pickle.dump(\n        checkpoint,\n        f,\n        protocol=pickle.HIGHEST_PROTOCOL\n    )\n\nprint(\"\\n✅ FINAL RECOVERY COMPLETE\")\nprint(\n    \"Checkpoint safely saved:\"\n)\nprint(CHECKPOINT)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T04:52:12.511486Z","iopub.execute_input":"2026-08-20T04:52:12.511834Z","iopub.status.idle":"2026-08-20T04:52:58.533235Z","shell.execute_reply.started":"2026-08-20T04:52:12.511803Z","shell.execute_reply":"2026-08-20T04:52:58.531972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\nBASE = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\n\ntrain = pd.read_csv(os.path.join(BASE, \"train.csv\"))\ntrain_series = pd.read_csv(os.path.join(BASE, \"train_series.csv\"))\n\nLABEL_COLS = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\",\n    \"Lateral Meniscus\", \"Medial OA\",\n    \"Lateral OA\", \"PF OA\", \"Effusion\",\n    \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\"\n]\n\nlabeled = train[\n    train[LABEL_COLS].notna().all(axis=1)\n].copy()\n\nstudy_ids = labeled[\"StudyInstanceUID\"].tolist()\n\nprint(\"Labeled studies:\", len(study_ids))\n\nprint(\"\\n=== SERIES COUNTS FROM CSV ===\")\n\ncounts = (\n    train_series[\n        train_series[\"StudyInstanceUID\"].isin(study_ids)\n    ]\n    .groupby(\"StudyInstanceUID\")\n    .size()\n)\n\nprint(counts.value_counts().sort_index())\n\nprint(\"\\n=== FIRST 10 STUDIES ===\")\n\nfor sid in study_ids[:10]:\n\n    rows = train_series[\n        train_series[\"StudyInstanceUID\"] == sid\n    ]\n\n    print(\n        \"\\nStudy:\",\n        sid\n    )\n\n    print(\n        rows[\n            [\n                \"SeriesInstanceUID\",\n                \"Fluid_Sensitive\",\n                \"Fat_Suppression\",\n                \"Anatomical_Plane\"\n            ]\n        ].to_string(index=False)\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T04:54:25.270345Z","iopub.execute_input":"2026-08-20T04:54:25.270716Z","iopub.status.idle":"2026-08-20T04:54:25.478106Z","shell.execute_reply.started":"2026-08-20T04:54:25.27068Z","shell.execute_reply":"2026-08-20T04:54:25.477353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(\"=== /kaggle/working ===\")\n\nfor f in os.listdir(\"/kaggle/working\"):\n    path = os.path.join(\"/kaggle/working\", f)\n    print(f, \"|\", round(os.path.getsize(path) / 1024**2, 2), \"MB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-14T04:45:37.836661Z","iopub.execute_input":"2026-08-14T04:45:37.83707Z","iopub.status.idle":"2026-08-14T04:45:37.841536Z","shell.execute_reply.started":"2026-08-14T04:45:37.837048Z","shell.execute_reply":"2026-08-14T04:45:37.841012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.multiclass import OneVsRestClassifier\nfrom sklearn.model_selection import LeaveOneOut\nfrom sklearn.metrics import roc_auc_score\n\n# Current champion representation\nX_weighted = 0.7 * X_mean + 0.3 * X_max\n\nprint(\"Weighted MRI shape:\", X_weighted.shape)\n\ndef evaluate_weighted_loo(X, Y, C=1.0):\n    loo = LeaveOneOut()\n    predictions = np.zeros_like(Y, dtype=float)\n\n    for train_idx, test_idx in loo.split(X):\n\n        model = make_pipeline(\n            StandardScaler(),\n            OneVsRestClassifier(\n                LogisticRegression(\n                    C=C,\n                    max_iter=3000,\n                    class_weight=\"balanced\"\n                )\n            )\n        )\n\n        model.fit(X[train_idx], Y[train_idx])\n\n        predictions[test_idx] = model.predict_proba(\n            X[test_idx]\n        )[0]\n\n    aucs = {}\n\n    for j, label in enumerate(label_cols):\n        if len(np.unique(Y[:, j])) < 2:\n            aucs[label] = np.nan\n        else:\n            aucs[label] = roc_auc_score(\n                Y[:, j],\n                predictions[:, j]\n            )\n\n    macro = np.nanmean(list(aucs.values()))\n\n    return aucs, macro, predictions\n\n\naucs, macro, weighted_pred = evaluate_weighted_loo(\n    X_weighted,\n    Y,\n    C=1.0\n)\n\nprint(\"\\nWeighted Mean/Max MRI\")\nprint(\"=\" * 50)\n\nfor label, auc in aucs.items():\n    print(f\"{label:20s} {auc:.4f}\")\n\nprint(f\"\\nMacro AUC: {macro:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T14:12:45.733784Z","iopub.execute_input":"2026-08-09T14:12:45.733952Z","iopub.status.idle":"2026-08-09T14:12:48.453111Z","shell.execute_reply.started":"2026-08-09T14:12:45.733934Z","shell.execute_reply":"2026-08-09T14:12:48.45211Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# FINAL SAFE RECOVERY + CHAMPION EVALUATION\n# ONE CELL\n# ============================================================\n\nimport os\nimport pickle\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torchvision.models as models\n\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import LeaveOneOut\nfrom sklearn.metrics import roc_auc_score\n\n# ------------------------------------------------------------\n# CONFIG\n# ------------------------------------------------------------\n\nBASE = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\nWORK = \"/kaggle/working\"\n\nMRI_CACHE = os.path.join(WORK, \"all_mri_data_variable.pkl\")\nEMB_CACHE = os.path.join(WORK, \"variable_series_embeddings.pkl\")\nFINAL_CACHE = os.path.join(WORK, \"FINAL_CHAMPION_CHECKPOINT.pkl\")\n\nLABEL_COLS = [\n    \"ACL\",\n    \"MCL\",\n    \"Medial Meniscus\",\n    \"Lateral Meniscus\",\n    \"Medial OA\",\n    \"Lateral OA\",\n    \"PF OA\",\n    \"Effusion\",\n    \"Synovitis\",\n    \"Baker's\",\n    \"Contusion\",\n    \"Fracture\"\n]\n\nprint(\"=\" * 65)\nprint(\"FINAL SAFE RECOVERY\")\nprint(\"=\" * 65)\n\n# ------------------------------------------------------------\n# 1. VERIFY ORIGINAL DATA\n# ------------------------------------------------------------\n\ntrain_path = os.path.join(BASE, \"train.csv\")\ntrain_series_path = os.path.join(BASE, \"train_series.csv\")\ntest_path = os.path.join(BASE, \"test.csv\")\ntest_series_path = os.path.join(BASE, \"test_series.csv\")\n\nassert os.path.exists(train_path), \"train.csv missing\"\nassert os.path.exists(train_series_path), \"train_series.csv missing\"\nassert os.path.exists(test_path), \"test.csv missing\"\nassert os.path.exists(test_series_path), \"test_series.csv missing\"\n\ntrain = pd.read_csv(train_path)\ntrain_series = pd.read_csv(train_series_path)\ntest = pd.read_csv(test_path)\ntest_series = pd.read_csv(test_series_path)\n\nprint(\"Train:\", train.shape)\nprint(\"Train series:\", train_series.shape)\nprint(\"Test:\", test.shape)\nprint(\"Test series:\", test_series.shape)\n\n# ------------------------------------------------------------\n# 2. LOAD LABELS\n# ------------------------------------------------------------\n\nlabeled_train = train[\n    train[LABEL_COLS].notna().all(axis=1)\n].copy()\n\nlabeled_train = labeled_train.iloc[:58].copy()\n\nstudy_ids = labeled_train[\"StudyInstanceUID\"].tolist()\nY = labeled_train[LABEL_COLS].values.astype(np.float32)\n\nprint(\"\\nLabeled studies:\", len(study_ids))\nprint(\"Y:\", Y.shape)\n\n# ------------------------------------------------------------\n# 3. LOAD SURVIVING MRI CHECKPOINT\n# ------------------------------------------------------------\n\nassert os.path.exists(\n    MRI_CACHE\n), \"MRI checkpoint missing: \" + MRI_CACHE\n\nwith open(MRI_CACHE, \"rb\") as f:\n    mri_cache = pickle.load(f)\n\nall_mri_data = mri_cache[\"all_mri_data\"]\n\nprint(\"\\nMRI checkpoint loaded.\")\nprint(\"Studies:\", len(all_mri_data))\nprint(\n    \"Series counts:\",\n    [len(all_mri_data[s]) for s in study_ids]\n)\n\n# ------------------------------------------------------------\n# 4. LOAD SURVIVING VARIABLE-SERIES EMBEDDINGS\n# ------------------------------------------------------------\n\nassert os.path.exists(\n    EMB_CACHE\n), \"Embedding checkpoint missing: \" + EMB_CACHE\n\nwith open(EMB_CACHE, \"rb\") as f:\n    emb_cache = pickle.load(f)\n\nvariable_embeddings = emb_cache[\"study_embeddings\"]\n\nX_variable = np.stack([\n    variable_embeddings[sid]\n    for sid in study_ids\n]).astype(np.float32)\n\nprint(\"\\nVariable-series embeddings:\")\nprint(\"Shape:\", X_variable.shape)\nprint(\"Mean:\", X_variable.mean())\nprint(\"Std :\", X_variable.std())\n\n# ------------------------------------------------------------\n# 5. REBUILD RESNET\n# ------------------------------------------------------------\n\nprint(\"\\nLoading ResNet18...\")\n\nweights = models.ResNet18_Weights.DEFAULT\nresnet = models.resnet18(weights=weights)\nresnet.fc = torch.nn.Identity()\nresnet.eval()\n\ndevice = torch.device(\"cpu\")\nresnet = resnet.to(device)\n\nimagenet_mean = torch.tensor(\n    [0.485, 0.456, 0.406],\n    dtype=torch.float32\n).view(1, 3, 1, 1)\n\nimagenet_std = torch.tensor(\n    [0.229, 0.224, 0.225],\n    dtype=torch.float32\n).view(1, 3, 1, 1)\n\n# ------------------------------------------------------------\n# 6. RECONSTRUCT EXACT 5-SERIES REPRESENTATION\n#\n# The known champion used 5 series/study:\n# 0.7 Mean + 0.3 Max\n#\n# We derive the first 5 available series from the\n# already-saved MRI cache. NO DICOM READING.\n# ------------------------------------------------------------\n\nprint(\"\\nReconstructing 5-series representation...\")\n\nX_mean_list = []\nX_max_list = []\n\nfor i, sid in enumerate(study_ids):\n\n    series_list = all_mri_data[sid]\n\n    # Preserve the first five series, matching our\n    # original five-series representation.\n    selected_series = series_list[:5]\n\n    if len(selected_series) < 5:\n        raise ValueError(\n            f\"{sid} has only {len(selected_series)} \"\n            \"usable series; cannot reproduce 5-series champion.\"\n        )\n\n    series_embs = []\n\n    for series in selected_series:\n\n        images = torch.from_numpy(\n            np.asarray(series)\n        ).float()\n\n        # 5 grayscale slices -> RGB\n        images = images.unsqueeze(1).repeat(\n            1, 3, 1, 1\n        )\n\n        images = (\n            images - imagenet_mean\n        ) / imagenet_std\n\n        with torch.no_grad():\n            emb = resnet(images)\n\n        # Average slices\n        series_embedding = (\n            emb.mean(dim=0)\n            .cpu()\n            .numpy()\n        )\n\n        series_embs.append(series_embedding)\n\n    series_embs = np.stack(series_embs)\n\n    X_mean_list.append(\n        series_embs.mean(axis=0)\n    )\n\n    X_max_list.append(\n        series_embs.max(axis=0)\n    )\n\nX_mean = np.stack(X_mean_list).astype(np.float32)\nX_max = np.stack(X_max_list).astype(np.float32)\n\n# Known champion\nX_weighted = (\n    0.7 * X_mean +\n    0.3 * X_max\n).astype(np.float32)\n\nprint(\"X_mean    :\", X_mean.shape)\nprint(\"X_max     :\", X_max.shape)\nprint(\"X_weighted:\", X_weighted.shape)\n\n# ------------------------------------------------------------\n# 7. EVALUATE KNOWN CHAMPION\n# ------------------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 65)\nprint(\"CHAMPION EVALUATION\")\nprint(\"=\" * 65)\n\nX = X_weighted\n\npredictions = np.zeros(\n   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T13:49:49.142108Z","iopub.execute_input":"2026-08-19T13:49:49.142403Z","iopub.status.idle":"2026-08-19T13:49:49.163816Z","shell.execute_reply.started":"2026-08-19T13:49:49.142374Z","shell.execute_reply":"2026-08-19T13:49:49.162803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pickle\nimport numpy as np\nimport torch\nimport torchvision.models as models\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import LeaveOneOut\nfrom sklearn.metrics import roc_auc_score\n\nWORK = \"/kaggle/working\"\n\nMRI_FILE = os.path.join(\n    WORK, \"all_mri_data_variable.pkl\"\n)\n\nEMB_FILE = os.path.join(\n    WORK, \"variable_series_embeddings.pkl\"\n)\n\nFINAL_FILE = os.path.join(\n    WORK, \"FINAL_CHAMPION_CHECKPOINT.pkl\"\n)\n\nLABEL_COLS = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\",\n    \"Lateral Meniscus\", \"Medial OA\",\n    \"Lateral OA\", \"PF OA\", \"Effusion\",\n    \"Synovitis\", \"Baker's\",\n    \"Contusion\", \"Fracture\"\n]\n\nprint(\"=\" * 60)\nprint(\"SAFE CHAMPION RECOVERY\")\nprint(\"=\" * 60)\n\n# --------------------------------------------------\n# CHECK FILES\n# --------------------------------------------------\n\nassert os.path.exists(MRI_FILE), \"MRI checkpoint missing\"\nassert os.path.exists(EMB_FILE), \"Embedding checkpoint missing\"\n\nprint(\"MRI checkpoint:\", round(\n    os.path.getsize(MRI_FILE) / 1024**2, 2\n), \"MB\")\n\nprint(\"Embedding checkpoint:\", round(\n    os.path.getsize(EMB_FILE) / 1024**2, 2\n), \"MB\")\n\n# --------------------------------------------------\n# LOAD MRI CHECKPOINT\n# --------------------------------------------------\n\nwith open(MRI_FILE, \"rb\") as f:\n    cache = pickle.load(f)\n\nall_mri_data = cache[\"all_mri_data\"]\nstudy_ids = cache[\"study_ids\"]\nY = cache[\"Y\"]\n\nprint(\"\\nStudies:\", len(study_ids))\nprint(\"Y:\", Y.shape)\n\n# --------------------------------------------------\n# LOAD VARIABLE EMBEDDINGS\n# --------------------------------------------------\n\nwith open(EMB_FILE, \"rb\") as f:\n    emb_cache = pickle.load(f)\n\nvariable_embeddings = emb_cache[\"study_embeddings\"]\n\nX_variable = np.stack([\n    variable_embeddings[sid]\n    for sid in study_ids\n]).astype(np.float32)\n\nprint(\"Variable embeddings:\", X_variable.shape)\n\n# --------------------------------------------------\n# RESNET\n# --------------------------------------------------\n\nprint(\"\\nLoading ResNet18...\")\n\nresnet = models.resnet18(\n    weights=models.ResNet18_Weights.DEFAULT\n)\n\nresnet.fc = torch.nn.Identity()\nresnet.eval()\n\nmean = torch.tensor(\n    [0.485, 0.456, 0.406],\n    dtype=torch.float32\n).view(1, 3, 1, 1)\n\nstd = torch.tensor(\n    [0.229, 0.224, 0.225],\n    dtype=torch.float32\n).view(1, 3, 1, 1)\n\n# --------------------------------------------------\n# REBUILD 5-SERIES MEAN + MAX\n# --------------------------------------------------\n\nX_mean_list = []\nX_max_list = []\n\nprint(\"\\nRebuilding 5-series representation...\")\n\nfor n, sid in enumerate(study_ids):\n\n    series_list = all_mri_data[sid]\n\n    if len(series_list) < 5:\n        raise ValueError(\n            f\"{sid} has only {len(series_list)} series\"\n        )\n\n    embeddings = []\n\n    for series in series_list[:5]:\n\n        images = torch.from_numpy(\n            np.asarray(series)\n        ).float()\n\n        images = images.unsqueeze(1).repeat(\n            1, 3, 1, 1\n        )\n\n        images = (images - mean) / std\n\n        with torch.no_grad():\n            out = resnet(images)\n\n        series_embedding = (\n            out.mean(dim=0)\n            .numpy()\n        )\n\n        embeddings.append(series_embedding)\n\n    embeddings = np.stack(embeddings)\n\n    X_mean_list.append(\n        embeddings.mean(axis=0)\n    )\n\n    X_max_list.append(\n        embeddings.max(axis=0)\n    )\n\n    if (n + 1) % 10 == 0:\n        print(\n            f\"Processed {n + 1}/{len(study_ids)}\"\n        )\n\nX_mean = np.stack(\n    X_mean_list\n).astype(np.float32)\n\nX_max = np.stack(\n    X_max_list\n).astype(np.float32)\n\nX_weighted = (\n    0.7 * X_mean +\n    0.3 * X_max\n).astype(np.float32)\n\nprint(\"\\nRepresentations:\")\nprint(\"Mean    :\", X_mean.shape)\nprint(\"Max     :\", X_max.shape)\nprint(\"Weighted:\", X_weighted.shape)\n\n# --------------------------------------------------\n# EVALUATE\n# --------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 60)\nprint(\"CHAMPION EVALUATION\")\nprint(\"=\" * 60)\n\npredictions = np.zeros_like(Y)\n\nfor j, label in enumerate(LABEL_COLS):\n\n    y = Y[:, j]\n    pred = np.zeros(len(y))\n\n    for train_idx, test_idx in LeaveOneOut().split(\n        X_weighted\n    ):\n\n        model = make_pipeline(\n            StandardScaler(),\n            LogisticRegression(\n                C=0.01,\n                max_iter=5000,\n                class_weight=\"balanced\"\n            )\n        )\n\n        model.fit(\n            X_weighted[train_idx],\n            y[train_idx]\n        )\n\n        pred[test_idx] = model.predict_proba(\n            X_weighted[test_idx]\n        )[0, 1]\n\n    predictions[:, j] = pred\n\n    print(\n        f\"{label:18s}\",\n        f\"{roc_auc_score(y, pred):.4f}\"\n    )\n\nmacro_auc = float(np.mean([\n    roc_auc_score(\n        Y[:, j],\n        predictions[:, j]\n    )\n    for j in range(Y.shape[1])\n]))\n\nprint(\"\\nMACRO AUC:\", macro_auc)\n\n# --------------------------------------------------\n# SAVE\n# --------------------------------------------------\n\ncheckpoint = {\n    \"X_mean\": X_mean,\n    \"X_max\": X_max,\n    \"X_weighted\": X_weighted,\n    \"X_variable\": X_variable,\n    \"Y\": Y,\n    \"study_ids\": study_ids,\n    \"label_cols\": LABEL_COLS,\n    \"predictions\": predictions,\n    \"macro_auc\": macro_auc,\n    \"mean_weight\": 0.7,\n    \"max_weight\": 0.3,\n    \"C\": 0.01\n}\n\nwith open(FINAL_FILE, \"wb\") as f:\n    pickle.dump(\n        checkpoint,\n        f,\n        protocol=pickle.HIGHEST_PROTOCOL\n    )\n\nprint(\"\\n\" + \"=\" * 60)\nprint(\"✅ CHAMPION CHECKPOINT SAVED\")\nprint(\"=\" * 60)\nprint(FINAL_FILE)\nprint(\n    \"Size MB:\",\n    round(\n        os.path.getsize(FINAL_FILE) / 1024**2,\n        2\n    )\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T13:50:59.469372Z","iopub.execute_input":"2026-08-19T13:50:59.469694Z","iopub.status.idle":"2026-08-19T13:51:11.951228Z","shell.execute_reply.started":"2026-08-19T13:50:59.469666Z","shell.execute_reply":"2026-08-19T13:51:11.950154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# series_embeddings should be:\n# (58 studies, variable number of series represented as 5 embeddings)\nprint(\"series_embeddings:\", series_embeddings.shape)\n\n# If shape is (58, 5, 512)\nX_mean = series_embeddings.mean(axis=1)\nX_max = series_embeddings.max(axis=1)\n\n# Current best pooling\nX_weighted = 0.7 * X_mean + 0.3 * X_max\n\nprint(\"Mean:\", X_mean.shape)\nprint(\"Max :\", X_max.shape)\nprint(\"Weighted:\", X_weighted.shape)\n\nprint(\"\\nWeighted statistics:\")\nprint(\"Mean:\", X_weighted.mean())\nprint(\"Std :\", X_weighted.std())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T14:13:40.210408Z","iopub.execute_input":"2026-08-09T14:13:40.211295Z","iopub.status.idle":"2026-08-09T14:13:40.217272Z","shell.execute_reply.started":"2026-08-09T14:13:40.211268Z","shell.execute_reply":"2026-08-09T14:13:40.216314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport torch\n\nprint(\"all_mri_data:\", type(all_mri_data))\nprint(\"Number of studies:\", len(all_mri_data))\n\n# Recreate per-series ResNet embeddings\nseries_embeddings_list = []\n\nwith torch.no_grad():\n\n    for study_id, series_list in all_mri_data.items():\n\n        study_series_embeddings = []\n\n        for series in series_list:\n\n            images = torch.from_numpy(\n                np.asarray(series)\n            ).float()\n\n            # (5, 224, 224) -> (5, 3, 224, 224)\n            images = images.unsqueeze(1).repeat(1, 3, 1, 1)\n\n            # ImageNet normalization\n            images = (images - mean) / std\n\n            # ResNet features\n            emb = resnet(images)\n\n            # Average slices\n            series_emb = emb.mean(dim=0).cpu().numpy()\n\n            study_series_embeddings.append(series_emb)\n\n        series_embeddings_list.append(\n            np.stack(study_series_embeddings)\n        )\n\nprint(\"Studies processed:\", len(series_embeddings_list))\nprint(\"Series counts:\", [x.shape[0] for x in series_embeddings_list])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T14:14:17.959317Z","iopub.execute_input":"2026-08-09T14:14:17.959589Z","iopub.status.idle":"2026-08-09T14:14:20.547819Z","shell.execute_reply.started":"2026-08-09T14:14:17.959566Z","shell.execute_reply":"2026-08-09T14:14:20.547035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Pool ALL available series for each study\n\nX_mean = np.stack([\n    x.mean(axis=0)\n    for x in series_embeddings_list\n])\n\nX_max = np.stack([\n    x.max(axis=0)\n    for x in series_embeddings_list\n])\n\nX_weighted = (\n    0.7 * X_mean +\n    0.3 * X_max\n)\n\nprint(\"Mean:\", X_mean.shape)\nprint(\"Max :\", X_max.shape)\nprint(\"Weighted:\", X_weighted.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T14:14:35.956112Z","iopub.execute_input":"2026-08-09T14:14:35.956372Z","iopub.status.idle":"2026-08-09T14:14:35.963296Z","shell.execute_reply.started":"2026-08-09T14:14:35.95635Z","shell.execute_reply":"2026-08-09T14:14:35.962471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport torch\n\n# ---------------------------------------------------------\n# Rebuild per-series ResNet embeddings from all_mri_data\n# ---------------------------------------------------------\n\nprint(\"Studies:\", len(all_mri_data))\n\nseries_embeddings_list = []\nstudy_ids_embedding = []\n\nresnet.eval()\n\nwith torch.no_grad():\n\n    for study_id, series_list in all_mri_data.items():\n\n        study_series_embeddings = []\n\n        for series in series_list:\n\n            # series: (5, 224, 224)\n            images = torch.from_numpy(\n                np.asarray(series)\n            ).float()\n\n            # (5,224,224) -> (5,3,224,224)\n            images = images.unsqueeze(1).repeat(1, 3, 1, 1)\n\n            # ImageNet normalization\n            images = (images - mean) / std\n\n            # ResNet embedding\n            emb = resnet(images)\n\n            # Average the 5 slices\n            series_emb = emb.mean(dim=0).cpu().numpy()\n\n            study_series_embeddings.append(series_emb)\n\n        series_embeddings_list.append(\n            np.stack(study_series_embeddings)\n        )\n\n        study_ids_embedding.append(study_id)\n\n\n# ---------------------------------------------------------\n# Verify\n# ---------------------------------------------------------\n\nprint(\"\\nStudies processed:\", len(series_embeddings_list))\nprint(\n    \"Series counts:\",\n    [x.shape[0] for x in series_embeddings_list]\n)\nprint(\n    \"Embedding dimensions:\",\n    set(x.shape[1] for x in series_embeddings_list)\n)\n\n# ---------------------------------------------------------\n# Pool ALL available series\n# ---------------------------------------------------------\n\nX_mean = np.stack([\n    x.mean(axis=0)\n    for x in series_embeddings_list\n])\n\nX_max = np.stack([\n    x.max(axis=0)\n    for x in series_embeddings_list\n])\n\n# Current champion weighting\nX_weighted = (\n    0.7 * X_mean +\n    0.3 * X_max\n)\n\nprint(\"\\nMean:\", X_mean.shape)\nprint(\"Max :\", X_max.shape)\nprint(\"Weighted:\", X_weighted.shape)\n\nprint(\"\\nWeighted statistics:\")\nprint(\"Mean:\", X_weighted.mean())\nprint(\"Std :\", X_weighted.std())\nprint(\"Min :\", X_weighted.min())\nprint(\"Max :\", X_weighted.max())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T14:15:19.298357Z","iopub.execute_input":"2026-08-09T14:15:19.298653Z","iopub.status.idle":"2026-08-09T14:15:19.309362Z","shell.execute_reply.started":"2026-08-09T14:15:19.298627Z","shell.execute_reply":"2026-08-09T14:15:19.308606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"j","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T14:16:31.540077Z","iopub.execute_input":"2026-08-09T14:16:31.540359Z","iopub.status.idle":"2026-08-09T14:16:31.546248Z","shell.execute_reply.started":"2026-08-09T14:16:31.540336Z","shell.execute_reply":"2026-08-09T14:16:31.54538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(\"/kaggle/working/mri_58_cache.pkl\", \"wb\") as f:\n    pickle.dump(all_mri_data, f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T14:17:47.681048Z","iopub.execute_input":"2026-08-09T14:17:47.681318Z","iopub.status.idle":"2026-08-09T14:17:47.688188Z","shell.execute_reply.started":"2026-08-09T14:17:47.681295Z","shell.execute_reply":"2026-08-09T14:17:47.687179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\nimport numpy as np\n\nprint(\"all_mri_data exists:\", \"all_mri_data\" in globals())\n\nif \"all_mri_data\" in globals():\n    print(\"Studies:\", len(all_mri_data))\nelse:\n    print(\"all_mri_data is NOT loaded.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T14:18:17.065467Z","iopub.execute_input":"2026-08-09T14:18:17.065774Z","iopub.status.idle":"2026-08-09T14:18:17.070901Z","shell.execute_reply.started":"2026-08-09T14:18:17.065744Z","shell.execute_reply":"2026-08-09T14:18:17.06997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(os.listdir(\"/kaggle/input/competitions\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-08T15:06:50.399377Z","iopub.execute_input":"2026-08-08T15:06:50.400101Z","iopub.status.idle":"2026-08-08T15:06:50.405257Z","shell.execute_reply.started":"2026-08-08T15:06:50.400066Z","shell.execute_reply":"2026-08-08T15:06:50.404364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a compact table for the 58 labeled studies\n\nreport_label_df = labeled_train[\n    [\"StudyInstanceUID\", \"Report\"] + target_cols\n].copy()\n\nprint(\"Rows:\", len(report_label_df))\nprint(\"Columns:\", report_label_df.shape[1])\n\ndisplay(\n    report_label_df[\n        [\"StudyInstanceUID\"] + target_cols\n    ].head()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:10:27.872798Z","iopub.execute_input":"2026-08-09T08:10:27.873211Z","iopub.status.idle":"2026-08-09T08:10:27.902161Z","shell.execute_reply.started":"2026-08-09T08:10:27.873181Z","shell.execute_reply":"2026-08-09T08:10:27.90107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Show report lengths\n\nreport_label_df[\"report_chars\"] = (\n    report_label_df[\"Report\"]\n    .fillna(\"\")\n    .astype(str)\n    .str.len()\n)\n\nprint(report_label_df[\"report_chars\"].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:11:10.224883Z","iopub.execute_input":"2026-08-09T08:11:10.225264Z","iopub.status.idle":"2026-08-09T08:11:10.243112Z","shell.execute_reply.started":"2026-08-09T08:11:10.225235Z","shell.execute_reply":"2026-08-09T08:11:10.242013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import roc_auc_score\nimport numpy as np\n\nX_text = report_label_df[\"Report\"].fillna(\"\").astype(str)\nY = report_label_df[target_cols].astype(int).values\n\n# TF-IDF works well as a first text baseline\nvectorizer = TfidfVectorizer(\n    lowercase=True,\n    ngram_range=(1, 2),\n    min_df=1,\n    max_features=20000,\n    sublinear_tf=True\n)\n\nX = vectorizer.fit_transform(X_text)\n\nprint(\"TF-IDF shape:\", X.shape)\nprint(\"Target shape:\", Y.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:12:16.926173Z","iopub.execute_input":"2026-08-09T08:12:16.926616Z","iopub.status.idle":"2026-08-09T08:12:21.212778Z","shell.execute_reply.started":"2026-08-09T08:12:16.926559Z","shell.execute_reply":"2026-08-09T08:12:21.210864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.base import clone\nimport numpy as np\n\nX_text = report_label_df[\"Report\"].fillna(\"\").astype(str).values\nY = report_label_df[target_cols].astype(int).values\n\nn = len(X_text)\npred = np.zeros_like(Y, dtype=float)\n\nfor i in range(n):\n    train_idx = np.arange(n) != i\n\n    vectorizer = TfidfVectorizer(\n        lowercase=True,\n        ngram_range=(1, 2),\n        min_df=1,\n        max_features=20000,\n        sublinear_tf=True\n    )\n\n    X_train = vectorizer.fit_transform(X_text[train_idx])\n    X_val = vectorizer.transform([X_text[i]])\n\n    for j, target in enumerate(target_cols):\n\n        y_train = Y[train_idx, j]\n\n        # Need both classes in the training fold\n        if len(np.unique(y_train)) < 2:\n            pred[i, j] = y_train.mean()\n            continue\n\n        model = LogisticRegression(\n            max_iter=2000,\n            class_weight=\"balanced\"\n        )\n\n        model.fit(X_train, y_train)\n        pred[i, j] = model.predict_proba(X_val)[0, 1]\n\nprint(\"Predictions shape:\", pred.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:13:00.397504Z","iopub.execute_input":"2026-08-09T08:13:00.398101Z","iopub.status.idle":"2026-08-09T08:13:13.700044Z","shell.execute_reply.started":"2026-08-09T08:13:00.398067Z","shell.execute_reply":"2026-08-09T08:13:13.698675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"auc_scores = {}\n\nfor j, target in enumerate(target_cols):\n    auc_scores[target] = roc_auc_score(Y[:, j], pred[:, j])\n\nprint(\"\\nPer-target AUC:\")\nfor target, score in auc_scores.items():\n    print(f\"{target:20s} {score:.4f}\")\n\nprint(\"\\nMacro AUC:\",\n      roc_auc_score(Y, pred, average=\"macro\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:14:13.575915Z","iopub.execute_input":"2026-08-09T08:14:13.576259Z","iopub.status.idle":"2026-08-09T08:14:13.61835Z","shell.execute_reply.started":"2026-08-09T08:14:13.576233Z","shell.execute_reply":"2026-08-09T08:14:13.617213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pydicom\nfrom PIL import Image\n\n# --------------------------------------------------\n# Configuration\n# --------------------------------------------------\n\nDATA = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\n\nIMG_SIZE = 224\nSLICES_PER_SERIES = 5\n\n# The 58 ground-truth studies\nlabeled_ids = labeled_train[\"StudyInstanceUID\"].tolist()\n\n# --------------------------------------------------\n# Helper: load representative slices from one series\n# --------------------------------------------------\n\ndef load_series_images(study_id, series_id, n_slices=5, img_size=224):\n    \n    series_path = os.path.join(\n        DATA,\n        \"train_series\",\n        study_id,\n        series_id\n    )\n    \n    files = [\n        os.path.join(series_path, f)\n        for f in os.listdir(series_path)\n        if f.endswith(\".dcm\")\n    ]\n    \n    if len(files) == 0:\n        return None\n    \n    # Read DICOMs\n    slices = []\n    \n    for f in files:\n        try:\n            ds = pydicom.dcmread(f)\n            slices.append(ds)\n        except Exception:\n            continue\n    \n    if len(slices) == 0:\n        return None\n    \n    # Sort by physical position when available\n    try:\n        slices.sort(\n            key=lambda x: float(x.ImagePositionPatient[2])\n        )\n    except Exception:\n        try:\n            slices.sort(\n                key=lambda x: int(x.InstanceNumber)\n            )\n        except Exception:\n            pass\n    \n    # Select evenly spaced slices\n    indices = np.linspace(\n        0,\n        len(slices) - 1,\n        min(n_slices, len(slices))\n    ).astype(int)\n    \n    selected = [slices[i] for i in indices]\n    \n    images = []\n    \n    for ds in selected:\n        \n        img = ds.pixel_array.astype(np.float32)\n        \n        # Robust normalization\n        p1, p99 = np.percentile(img, [1, 99])\n        \n        if p99 > p1:\n            img = np.clip(img, p1, p99)\n            img = (img - p1) / (p99 - p1)\n        else:\n            img = np.zeros_like(img)\n        \n        # Resize\n        img = Image.fromarray(img)\n        img = img.resize(\n            (img_size, img_size),\n            Image.Resampling.BILINEAR\n        )\n        \n        images.append(np.asarray(img, dtype=np.float32))\n    \n    # If fewer than requested slices, repeat the final slice\n    while len(images) < n_slices:\n        images.append(images[-1].copy())\n    \n    return np.stack(images)\n\n\n# --------------------------------------------------\n# Build MRI tensor for one study\n# --------------------------------------------------\n\ndef build_study_tensor(study_id):\n    \n    study_rows = train_series[\n        train_series[\"StudyInstanceUID\"] == study_id\n    ].copy()\n    \n    series_tensors = []\n    \n    # Prefer fluid-sensitive series\n    study_rows = study_rows.sort_values(\n        [\"Fluid_Sensitive\", \"Fat_Suppression\"],\n        ascending=False\n    )\n    \n    for _, row in study_rows.iterrows():\n        \n        tensor = load_series_images(\n            study_id,\n            row[\"SeriesInstanceUID\"],\n            n_slices=SLICES_PER_SERIES,\n            img_size=IMG_SIZE\n        )\n        \n        if tensor is not None:\n            series_tensors.append(tensor)\n    \n    return series_tensors","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:15:39.110364Z","iopub.execute_input":"2026-08-09T08:15:39.110715Z","iopub.status.idle":"2026-08-09T08:15:39.124613Z","shell.execute_reply.started":"2026-08-09T08:15:39.110688Z","shell.execute_reply":"2026-08-09T08:15:39.123832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_id = labeled_ids[0]\n\nseries_tensors = build_study_tensor(test_id)\n\nprint(\"Study:\", test_id)\nprint(\"Number of usable series:\", len(series_tensors))\n\nfor i, x in enumerate(series_tensors):\n    print(i, x.shape, x.min(), x.max(), x.mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:16:00.737161Z","iopub.execute_input":"2026-08-09T08:16:00.73754Z","iopub.status.idle":"2026-08-09T08:16:02.120758Z","shell.execute_reply.started":"2026-08-09T08:16:00.737512Z","shell.execute_reply":"2026-08-09T08:16:02.119729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pickle\nfrom tqdm.auto import tqdm\n\nMRI_CACHE = \"/kaggle/working/mri_58_cache.pkl\"\n\nall_mri_data = {}\n\nfor study_id in tqdm(labeled_ids, desc=\"Building MRI cache\"):\n    try:\n        series_tensors = build_study_tensor(study_id)\n\n        all_mri_data[study_id] = series_tensors\n\n    except Exception as e:\n        print(f\"\\nERROR: {study_id}\")\n        print(e)\n        all_mri_data[study_id] = None\n\nwith open(MRI_CACHE, \"wb\") as f:\n    pickle.dump(all_mri_data, f)\n\nprint(\"\\nSaved:\", MRI_CACHE)\nprint(\"Studies:\", len(all_mri_data))\n\nsuccessful = sum(\n    v is not None and len(v) > 0\n    for v in all_mri_data.values()\n)\n\nprint(\"Successful studies:\", successful)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:16:50.868425Z","iopub.execute_input":"2026-08-09T08:16:50.868838Z","iopub.status.idle":"2026-08-09T08:18:42.433452Z","shell.execute_reply.started":"2026-08-09T08:16:50.868806Z","shell.execute_reply":"2026-08-09T08:18:42.43245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lengths = [\n    len(v) for v in all_mri_data.values()\n    if v is not None\n]\n\nprint(\"Number of series per study:\")\nprint(lengths)\n\nprint(\"\\nMinimum series:\", min(lengths))\nprint(\"Maximum series:\", max(lengths))\nprint(\"Average series:\", np.mean(lengths))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:19:39.435969Z","iopub.execute_input":"2026-08-09T08:19:39.436346Z","iopub.status.idle":"2026-08-09T08:19:39.444052Z","shell.execute_reply.started":"2026-08-09T08:19:39.436316Z","shell.execute_reply":"2026-08-09T08:19:39.442715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\nimport numpy as np\n\nwith open(\"/kaggle/working/mri_58_cache.pkl\", \"rb\") as f:\n    all_mri_data = pickle.load(f)\n\nN_SERIES = 5\nN_SLICES = 5\nIMG_SIZE = 224\n\nX_mri = []\n\nfor study_id in labeled_ids:\n\n    series_list = all_mri_data[study_id]\n\n    # Keep 5 series\n    selected = series_list[:N_SERIES]\n\n    # If fewer than 5, repeat the last available series\n    while len(selected) < N_SERIES:\n        selected.append(selected[-1])\n\n    study_tensor = np.stack(selected)\n\n    X_mri.append(study_tensor)\n\nX_mri = np.stack(X_mri).astype(np.float32)\n\nprint(\"MRI tensor shape:\", X_mri.shape)\nprint(\"Min:\", X_mri.min())\nprint(\"Max:\", X_mri.max())\nprint(\"Mean:\", X_mri.mean())\nprint(\"Std:\", X_mri.std())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:20:48.547531Z","iopub.execute_input":"2026-08-09T08:20:48.548665Z","iopub.status.idle":"2026-08-09T08:20:49.341215Z","shell.execute_reply.started":"2026-08-09T08:20:48.548616Z","shell.execute_reply":"2026-08-09T08:20:49.339963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\ndef extract_mri_features(study_tensor):\n    \"\"\"\n    study_tensor:\n        (5 series, 5 slices, 224, 224)\n    \"\"\"\n\n    features = []\n\n    # Overall intensity statistics\n    flat = study_tensor.reshape(-1)\n\n    features.extend([\n        np.mean(flat),\n        np.std(flat),\n        np.min(flat),\n        np.max(flat),\n        np.percentile(flat, 1),\n        np.percentile(flat, 5),\n        np.percentile(flat, 25),\n        np.percentile(flat, 50),\n        np.percentile(flat, 75),\n        np.percentile(flat, 95),\n        np.percentile(flat, 99),\n    ])\n\n    # Statistics for each series\n    for s in range(study_tensor.shape[0]):\n        x = study_tensor[s].reshape(-1)\n\n        features.extend([\n            np.mean(x),\n            np.std(x),\n            np.percentile(x, 25),\n            np.percentile(x, 50),\n            np.percentile(x, 75),\n        ])\n\n    # Statistics for each slice\n    for s in range(study_tensor.shape[0]):\n        for k in range(study_tensor.shape[1]):\n            x = study_tensor[s, k]\n\n            features.extend([\n                np.mean(x),\n                np.std(x),\n            ])\n\n    return np.asarray(features, dtype=np.float32)\n\n\nX_mri_features = np.stack([\n    extract_mri_features(all_mri_data[study_id][:5] \n                         if len(all_mri_data[study_id]) >= 5\n                         else all_mri_data[study_id])\n    for study_id in labeled_ids\n])\n\nprint(\"MRI feature shape:\", X_mri_features.shape)\nprint(\"Feature min:\", X_mri_features.min())\nprint(\"Feature max:\", X_mri_features.max())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:22:02.773911Z","iopub.execute_input":"2026-08-09T08:22:02.774303Z","iopub.status.idle":"2026-08-09T08:22:02.791698Z","shell.execute_reply.started":"2026-08-09T08:22:02.774205Z","shell.execute_reply":"2026-08-09T08:22:02.790303Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Y = report_label_df[label_cols].values.astype(float)\n\nresults = {}\n\nfor name, X in [\n    (\"Mean\", X_mean),\n    (\"Max\", X_max),\n    (\"Mean+Max\", X_meanmax),\n]:\n    aucs, macro, pred = evaluate_embedding_loo(X, Y)\n\n    results[name] = macro\n\n    print(\"\\n\" + \"=\"*50)\n    print(name)\n    print(\"=\"*50)\n\n    for label, auc in aucs.items():\n        print(f\"{label:20s} {auc:.4f}\")\n\n    print(f\"\\nMacro AUC: {macro:.4f}\")\n\nprint(\"\\nFINAL:\")\nfor name, score in results.items():\n    print(f\"{name:10s}: {score:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T10:11:50.589397Z","iopub.execute_input":"2026-08-09T10:11:50.590524Z","iopub.status.idle":"2026-08-09T10:12:27.257914Z","shell.execute_reply.started":"2026-08-09T10:11:50.590473Z","shell.execute_reply":"2026-08-09T10:12:27.257063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for name, X in [\n    (\"Max\", X_max),\n    (\"Mean+Max\", X_meanmax),\n]:\n    \n    aucs, macro, pred = evaluate_embedding_loo(X, Y)\n\n    print(\"\\n\" + \"=\" * 50)\n    print(name)\n    print(\"=\" * 50)\n\n    for label, auc in aucs.items():\n        print(f\"{label:20s} {auc:.4f}\")\n\n    print(f\"\\nMacro AUC: {macro:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T10:12:47.307553Z","iopub.execute_input":"2026-08-09T10:12:47.30794Z","iopub.status.idle":"2026-08-09T10:13:12.18729Z","shell.execute_reply.started":"2026-08-09T10:12:47.307908Z","shell.execute_reply":"2026-08-09T10:13:12.185658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"weights = [\n    0.0,\n    0.1,\n    0.2,\n    0.3,\n    0.4,\n    0.5,\n    0.6,\n    0.7,\n    0.8,\n    0.9,\n    1.0\n]\n\nweighted_results = {}\n\nfor alpha in weights:\n\n    X_weighted = (\n        alpha * X_mean +\n        (1 - alpha) * X_max\n    )\n\n    aucs, macro, pred = evaluate_embedding_loo(\n        X_weighted,\n        Y\n    )\n\n    weighted_results[alpha] = macro\n\n    print(\n        f\"Mean weight: {alpha:.1f} | \"\n        f\"Max weight: {1-alpha:.1f} | \"\n        f\"Macro AUC: {macro:.4f}\"\n    )\n\nbest_alpha = max(\n    weighted_results,\n    key=weighted_results.get\n)\n\nprint(\"\\n\" + \"=\" * 60)\nprint(\"BEST WEIGHTED POOLING\")\nprint(\"=\" * 60)\nprint(\"Mean weight:\", best_alpha)\nprint(\"Max weight:\", 1 - best_alpha)\nprint(\"Macro AUC:\", weighted_results[best_alpha])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T10:14:36.603125Z","iopub.execute_input":"2026-08-09T10:14:36.603455Z","iopub.status.idle":"2026-08-09T10:16:31.359654Z","shell.execute_reply.started":"2026-08-09T10:14:36.603431Z","shell.execute_reply":"2026-08-09T10:16:31.358842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\n\nprint(\"=== WORKING DIRECTORY ===\")\nprint(os.getcwd())\n\nprint(\"\\n=== FILES IN /kaggle/working ===\")\nfor p in glob.glob(\"/kaggle/working/**/*\", recursive=True):\n    print(p)\n\nprint(\"\\n=== NOTEBOOK FILES ===\")\nfor root in [\"/kaggle/working\", \"/kaggle/input\"]:\n    for p in glob.glob(root + \"/**/*.ipynb\", recursive=True):\n        print(p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T03:31:54.741633Z","iopub.execute_input":"2026-08-12T03:31:54.742067Z","iopub.status.idle":"2026-08-12T03:39:34.294815Z","shell.execute_reply.started":"2026-08-12T03:31:54.742012Z","shell.execute_reply":"2026-08-12T03:39:34.291947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nfor path in [\"/kaggle/input\", \"/kaggle/input/rsna\", \"/kaggle/input/mri\"]:\n    print(\"\\n\", path)\n    if os.path.exists(path):\n        print(os.listdir(path)[:20])\n    else:\n        print(\"Does not exist\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T03:39:34.621041Z","iopub.execute_input":"2026-08-12T03:39:34.621503Z","iopub.status.idle":"2026-08-12T03:39:34.631181Z","shell.execute_reply.started":"2026-08-12T03:39:34.621472Z","shell.execute_reply":"2026-08-12T03:39:34.629021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\n\nprint(\"Kaggle input folders:\")\nfor x in glob.glob(\"/kaggle/input/*\"):\n    print(x)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T03:39:34.633726Z","iopub.execute_input":"2026-08-12T03:39:34.634445Z","iopub.status.idle":"2026-08-12T03:39:34.670446Z","shell.execute_reply.started":"2026-08-12T03:39:34.634406Z","shell.execute_reply":"2026-08-12T03:39:34.668936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nroot = \"/kaggle/input/competitions\"\n\nfor dirpath, dirnames, filenames in os.walk(root):\n    level = dirpath.replace(root, \"\").count(os.sep)\n    \n    if level <= 2:\n        print(\"\\n\" + \"  \" * level + os.path.basename(dirpath) + \"/\")\n        for f in filenames[:20]:\n            print(\"  \" * (level + 1) + f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T03:40:56.233966Z","iopub.execute_input":"2026-08-12T03:40:56.234355Z","iopub.status.idle":"2026-08-12T03:46:48.340961Z","shell.execute_reply.started":"2026-08-12T03:40:56.234325Z","shell.execute_reply":"2026-08-12T03:46:48.339208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\nBASE = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\n\ntrain = pd.read_csv(os.path.join(BASE, \"train.csv\"))\ntrain_series = pd.read_csv(os.path.join(BASE, \"train_series.csv\"))\n\nprint(\"train shape:\", train.shape)\nprint(\"train_series shape:\", train_series.shape)\n\nprint(\"\\ntrain columns:\")\nprint(train.columns.tolist())\n\nprint(\"\\ntrain_series columns:\")\nprint(train_series.columns.tolist())\n\nprint(\"\\nFirst train_series rows:\")\ndisplay(train_series.head())\n\nprint(\"\\nTrain series directory exists:\",\n      os.path.exists(os.path.join(BASE, \"train_series\")))\n\nprint(\"Number of study folders:\",\n      len(os.listdir(os.path.join(BASE, \"train_series\"))))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T03:46:48.784056Z","iopub.execute_input":"2026-08-12T03:46:48.784408Z","iopub.status.idle":"2026-08-12T03:46:51.488786Z","shell.execute_reply.started":"2026-08-12T03:46:48.784379Z","shell.execute_reply":"2026-08-12T03:46:51.486168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_cols = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\",\n    \"Medial OA\", \"Lateral OA\", \"PF OA\", \"Effusion\",\n    \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\"\n]\n\n# Recreate the 58-study labeled set\nlabeled_train = train[\n    train[label_cols].notna().all(axis=1)\n].copy()\n\nprint(\"Labeled studies:\", len(labeled_train))\nprint(\"\\nPositive cases:\")\nprint(labeled_train[label_cols].sum())\n\nprint(\"\\nSeries available for these studies:\")\nsubset_series = train_series[\n    train_series[\"StudyInstanceUID\"].isin(\n        labeled_train[\"StudyInstanceUID\"]\n    )\n].copy()\n\nseries_counts = (\n    subset_series.groupby(\"StudyInstanceUID\")\n    .size()\n)\n\nprint(series_counts.tolist())\nprint(\"\\nMin series:\", series_counts.min())\nprint(\"Max series:\", series_counts.max())\nprint(\"Mean series:\", series_counts.mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T03:50:36.810356Z","iopub.execute_input":"2026-08-12T03:50:36.811432Z","iopub.status.idle":"2026-08-12T03:50:36.857604Z","shell.execute_reply.started":"2026-08-12T03:50:36.81139Z","shell.execute_reply":"2026-08-12T03:50:36.856206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nfrom tqdm.auto import tqdm\n\nBASE = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\nTRAIN_SERIES_DIR = os.path.join(BASE, \"train_series\")\n\n# Make sure these functions exist\nprint(\"load_series_images:\", \"load_series_images\" in globals())\nprint(\"build_study_tensor:\", \"build_study_tensor\" in globals())\nprint(\"normalize_mri:\", \"normalize_mri\" in globals())\nprint(\"resize_mri:\", \"resize_mri\" in globals())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T03:52:01.055978Z","iopub.execute_input":"2026-08-12T03:52:01.056509Z","iopub.status.idle":"2026-08-12T03:52:01.163294Z","shell.execute_reply.started":"2026-08-12T03:52:01.056472Z","shell.execute_reply":"2026-08-12T03:52:01.16178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import importlib.util\nimport os\n\n# ============================================================\n# 1. Check required libraries\n# ============================================================\n\npackages = [\n    \"pydicom\",\n    \"cv2\",\n    \"PIL\",\n    \"scipy\",\n    \"numpy\",\n    \"torch\",\n    \"torchvision\",\n    \"timm\"\n]\n\nprint(\"=== AVAILABLE LIBRARIES ===\")\nfor p in packages:\n    print(f\"{p:12s}:\", bool(importlib.util.find_spec(p)))\n\n\n# ============================================================\n# 2. Inspect the first labeled study\n# ============================================================\n\nstudy_id = labeled_train[\"StudyInstanceUID\"].iloc[0]\n\nstudy_path = os.path.join(\n    TRAIN_SERIES_DIR,\n    study_id\n)\n\nprint(\"\\n=== FIRST LABELED STUDY ===\")\nprint(\"Study:\", study_id)\nprint(\"Path:\", study_path)\nprint(\"Exists:\", os.path.exists(study_path))\n\nif os.path.exists(study_path):\n    series_folders = os.listdir(study_path)\n\n    print(\"Number of series:\", len(series_folders))\n    print(\"Series folders:\")\n\n    for i, series_id in enumerate(series_folders):\n        series_path = os.path.join(study_path, series_id)\n\n        if os.path.isdir(series_path):\n            files = os.listdir(series_path)\n\n            print(\n                f\"  {i}: {series_id} | \"\n                f\"files: {len(files)}\"\n            )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T03:53:01.776932Z","iopub.execute_input":"2026-08-12T03:53:01.777468Z","iopub.status.idle":"2026-08-12T03:53:01.861254Z","shell.execute_reply.started":"2026-08-12T03:53:01.777437Z","shell.execute_reply":"2026-08-12T03:53:01.86009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pydicom\n\n# Use the first series of the first labeled study\nseries_id = os.listdir(study_path)[0]\nseries_path = os.path.join(study_path, series_id)\n\nfiles = [\n    os.path.join(series_path, f)\n    for f in os.listdir(series_path)\n    if os.path.isfile(os.path.join(series_path, f))\n]\n\nprint(\"Series:\", series_id)\nprint(\"Number of files:\", len(files))\n\n# Read the first DICOM\nds = pydicom.dcmread(files[0], stop_before_pixels=False)\n\nprint(\"\\n=== DICOM INFORMATION ===\")\nprint(\"Rows:\", getattr(ds, \"Rows\", None))\nprint(\"Columns:\", getattr(ds, \"Columns\", None))\nprint(\"NumberOfFrames:\", getattr(ds, \"NumberOfFrames\", None))\nprint(\"SliceThickness:\", getattr(ds, \"SliceThickness\", None))\nprint(\"InstanceNumber:\", getattr(ds, \"InstanceNumber\", None))\nprint(\"ImagePositionPatient:\", getattr(ds, \"ImagePositionPatient\", None))\nprint(\"Pixel array shape:\", ds.pixel_array.shape)\nprint(\"Pixel dtype:\", ds.pixel_array.dtype)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T03:54:15.879729Z","iopub.execute_input":"2026-08-12T03:54:15.880455Z","iopub.status.idle":"2026-08-12T03:54:16.998327Z","shell.execute_reply.started":"2026-08-12T03:54:15.880424Z","shell.execute_reply":"2026-08-12T03:54:16.996911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pydicom\n\n# ------------------------------------------------------------\n# Read all DICOM slices from the first series\n# ------------------------------------------------------------\n\nrecords = []\n\nfor fname in os.listdir(series_path):\n\n    fpath = os.path.join(series_path, fname)\n\n    if not os.path.isfile(fpath):\n        continue\n\n    try:\n        ds = pydicom.dcmread(fpath)\n\n        pixel = ds.pixel_array.astype(np.float32)\n\n        ipp = getattr(ds, \"ImagePositionPatient\", None)\n        z = float(ipp[2]) if ipp is not None else np.nan\n\n        records.append({\n            \"file\": fname,\n            \"instance\": getattr(ds, \"InstanceNumber\", None),\n            \"z\": z,\n            \"mean\": pixel.mean(),\n            \"std\": pixel.std(),\n            \"min\": pixel.min(),\n            \"max\": pixel.max(),\n            \"shape\": pixel.shape\n        })\n\n    except Exception:\n        pass\n\n\n# ------------------------------------------------------------\n# Sort by physical slice position\n# ------------------------------------------------------------\n\nrecords = sorted(\n    records,\n    key=lambda x: (\n        np.inf if np.isnan(x[\"z\"]) else x[\"z\"]\n    )\n)\n\nprint(\"Number of readable slices:\", len(records))\n\nprint(\"\\nFirst 10 slices:\")\nfor r in records[:10]:\n    print(\n        f\"Instance={r['instance']:>3} | \"\n        f\"z={r['z']:>10.3f} | \"\n        f\"mean={r['mean']:>9.2f} | \"\n        f\"std={r['std']:>9.2f} | \"\n        f\"range=({r['min']:.0f}, {r['max']:.0f})\"\n    )\n\nprint(\"\\nLast 10 slices:\")\nfor r in records[-10:]:\n    print(\n        f\"Instance={r['instance']:>3} | \"\n        f\"z={r['z']:>10.3f} | \"\n        f\"mean={r['mean']:>9.2f} | \"\n        f\"std={r['std']:>9.2f} | \"\n        f\"range=({r['min']:.0f}, {r['max']:.0f})\"\n    )\n\nprint(\"\\nUnique image shapes:\")\nprint(set(r[\"shape\"] for r in records))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T03:55:10.791513Z","iopub.execute_input":"2026-08-12T03:55:10.792739Z","iopub.status.idle":"2026-08-12T03:55:11.225121Z","shell.execute_reply.started":"2026-08-12T03:55:10.792696Z","shell.execute_reply":"2026-08-12T03:55:11.223957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Slice indices produced by common 5-slice sampling strategies\nn = len(records)\n\nprint(\"Total slices:\", n)\n\nprint(\"\\nEvery 6th slice:\")\nprint(np.linspace(0, n - 1, 5, dtype=int))\n\nprint(\"\\nMiddle 5 consecutive slices:\")\nmid = n // 2\nprint(np.arange(mid - 2, mid + 3))\n\nprint(\"\\n5 evenly spaced positions:\")\nprint([\n    round(i * (n - 1) / 4)\n    for i in range(5)\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T03:56:01.530387Z","iopub.execute_input":"2026-08-12T03:56:01.531241Z","iopub.status.idle":"2026-08-12T03:56:01.539848Z","shell.execute_reply.started":"2026-08-12T03:56:01.531201Z","shell.execute_reply":"2026-08-12T03:56:01.53864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport cv2\nfrom tqdm.auto import tqdm\n\nBASE = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\nTRAIN_SERIES_DIR = os.path.join(BASE, \"train_series\")\n\n# ------------------------------------------------------------\n# MRI preprocessing\n# ------------------------------------------------------------\n\ndef normalize_mri(img):\n    img = img.astype(np.float32)\n\n    lo = np.percentile(img, 1)\n    hi = np.percentile(img, 99)\n\n    if hi <= lo:\n        return np.zeros_like(img, dtype=np.float32)\n\n    img = np.clip(img, lo, hi)\n    img = (img - lo) / (hi - lo)\n\n    return img.astype(np.float32)\n\n\ndef resize_mri(img, size=224):\n    return cv2.resize(\n        img,\n        (size, size),\n        interpolation=cv2.INTER_AREA\n    ).astype(np.float32)\n\n\ndef load_series_images(series_path, n_slices=5):\n    records = []\n\n    for fname in os.listdir(series_path):\n\n        fpath = os.path.join(series_path, fname)\n\n        if not os.path.isfile(fpath):\n            continue\n\n        try:\n            ds = pydicom.dcmread(fpath)\n\n            pixel = ds.pixel_array.astype(np.float32)\n\n            instance = getattr(ds, \"InstanceNumber\", None)\n\n            if instance is None:\n                instance = len(records)\n\n            records.append((int(instance), pixel))\n\n        except Exception:\n            continue\n\n    if len(records) == 0:\n        return None\n\n    # Sort slices anatomically/order-wise\n    records.sort(key=lambda x: x[0])\n\n    # Select 5 evenly spaced slices\n    indices = np.linspace(\n        0,\n        len(records) - 1,\n        n_slices,\n        dtype=int\n    )\n\n    selected = []\n\n    for idx in indices:\n        img = records[idx][1]\n        img = normalize_mri(img)\n        img = resize_mri(img, 224)\n        selected.append(img)\n\n    return np.stack(selected).astype(np.float32)\n\n\n# ------------------------------------------------------------\n# Build all_mri_data\n# ------------------------------------------------------------\n\nall_mri_data = {}\n\nstudy_ids = labeled_train[\"StudyInstanceUID\"].tolist()\n\nfor study_id in tqdm(study_ids, desc=\"Building MRI data\"):\n\n    study_path = os.path.join(\n        TRAIN_SERIES_DIR,\n        study_id\n    )\n\n    if not os.path.isdir(study_path):\n        continue\n\n    series_list = []\n\n    series_ids = sorted(os.listdir(study_path))\n\n    for series_id in series_ids:\n\n        series_path = os.path.join(\n            study_path,\n            series_id\n        )\n\n        if not os.path.isdir(series_path):\n            continue\n\n        series = load_series_images(\n            series_path,\n            n_slices=5\n        )\n\n        if series is not None and series.shape == (5, 224, 224):\n            series_list.append(series)\n\n    if len(series_list) == 0:\n        continue\n\n    # --------------------------------------------------------\n    # Keep exactly 5 series to match the previous tensor format\n    # --------------------------------------------------------\n\n    if len(series_list) >= 5:\n        selected_series = series_list[:5]\n\n    else:\n        selected_series = series_list.copy()\n\n        while len(selected_series) < 5:\n            selected_series.append(selected_series[-1].copy())\n\n    all_mri_data[study_id] = selected_series\n\n\n# ------------------------------------------------------------\n# Convert to tensor\n# ------------------------------------------------------------\n\nX_mri = np.stack([\n    np.stack(all_mri_data[sid])\n    for sid in study_ids\n    if sid in all_mri_data\n]).astype(np.float32)\n\nprint(\"\\n========================================\")\nprint(\"MRI reconstruction complete\")\nprint(\"========================================\")\nprint(\"Studies:\", len(all_mri_data))\nprint(\"X_mri shape:\", X_mri.shape)\nprint(\"Min:\", X_mri.min())\nprint(\"Max:\", X_mri.max())\nprint(\"Mean:\", X_mri.mean())\nprint(\"Std:\", X_mri.std())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T03:57:13.92233Z","iopub.execute_input":"2026-08-12T03:57:13.922711Z","iopub.status.idle":"2026-08-12T03:59:59.169567Z","shell.execute_reply.started":"2026-08-12T03:57:13.92268Z","shell.execute_reply":"2026-08-12T03:59:59.168132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\n\nwith open(\"/kaggle/working/mri_58_reconstructed.pkl\", \"wb\") as f:\n    pickle.dump(all_mri_data, f)\n\nprint(\"✅ MRI data saved!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T04:02:45.457528Z","iopub.execute_input":"2026-08-12T04:02:45.458199Z","iopub.status.idle":"2026-08-12T04:02:47.374657Z","shell.execute_reply.started":"2026-08-12T04:02:45.458121Z","shell.execute_reply":"2026-08-12T04:02:47.372385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\n\nwith open(\"/kaggle/working/mri_58_reconstructed.pkl\", \"rb\") as f:\n    all_mri_data = pickle.load(f)\n\nprint(\"✅ MRI data loaded!\")\nprint(\"Studies:\", len(all_mri_data))\nprint(\"First study series:\", len(next(iter(all_mri_data.values()))))\nprint(\"First series shape:\", next(iter(all_mri_data.values()))[0].shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T04:03:23.310631Z","iopub.execute_input":"2026-08-12T04:03:23.311328Z","iopub.status.idle":"2026-08-12T04:03:24.146612Z","shell.execute_reply.started":"2026-08-12T04:03:23.311286Z","shell.execute_reply":"2026-08-12T04:03:24.145059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\nstudy_ids = labeled_train[\"StudyInstanceUID\"].tolist()\n\nX_mri = np.stack([\n    np.stack(all_mri_data[sid])\n    for sid in study_ids\n]).astype(np.float32)\n\nprint(\"✅ X_mri rebuilt!\")\nprint(\"Shape:\", X_mri.shape)\nprint(\"Min:\", X_mri.min())\nprint(\"Max:\", X_mri.max())\nprint(\"Mean:\", X_mri.mean())\nprint(\"Std:\", X_mri.std())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T04:03:56.497742Z","iopub.execute_input":"2026-08-12T04:03:56.498859Z","iopub.status.idle":"2026-08-12T04:03:57.470201Z","shell.execute_reply.started":"2026-08-12T04:03:56.498715Z","shell.execute_reply":"2026-08-12T04:03:57.4655Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torchvision.models as models\n\ndevice = torch.device(\"cpu\")\n\n# Load pretrained ResNet18\nresnet = models.resnet18(weights=\"DEFAULT\")\n\n# Remove classification layer\nresnet.fc = torch.nn.Identity()\n\nresnet = resnet.to(device)\nresnet.eval()\n\n# ImageNet normalization\nmean = torch.tensor(\n    [0.485, 0.456, 0.406],\n    dtype=torch.float32\n).view(1, 3, 1, 1)\n\nstd = torch.tensor(\n    [0.229, 0.224, 0.225],\n    dtype=torch.float32\n).view(1, 3, 1, 1)\n\nprint(\"✅ ResNet18 rebuilt\")\nprint(\"Embedding dimension: 512\")\nprint(\"Device:\", device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T04:04:34.528384Z","iopub.execute_input":"2026-08-12T04:04:34.528948Z","iopub.status.idle":"2026-08-12T04:04:52.9116Z","shell.execute_reply.started":"2026-08-12T04:04:34.52891Z","shell.execute_reply":"2026-08-12T04:04:52.910075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport torch\nfrom tqdm.auto import tqdm\n\nseries_embeddings = []\n\nwith torch.no_grad():\n\n    for study_idx in tqdm(\n        range(X_mri.shape[0]),\n        desc=\"Extracting series embeddings\"\n    ):\n\n        study_series_embeddings = []\n\n        for series_idx in range(X_mri.shape[1]):\n\n            # (5, 224, 224)\n            series = X_mri[study_idx, series_idx]\n\n            # (5, 224, 224) -> (5, 3, 224, 224)\n            images = torch.from_numpy(series).float()\n            images = images.unsqueeze(1).repeat(1, 3, 1, 1)\n\n            # ImageNet normalization\n            images = (images - mean) / std\n\n            images = images.to(device)\n\n            # (5, 512)\n            embeddings = resnet(images)\n\n            # Average the 5 slices -> (512,)\n            series_embedding = embeddings.mean(dim=0)\n\n            study_series_embeddings.append(\n                series_embedding.cpu().numpy()\n            )\n\n        # (5, 512)\n        series_embeddings.append(\n            np.stack(study_series_embeddings)\n        )\n\nseries_embeddings = np.stack(series_embeddings).astype(np.float32)\n\nprint(\"\\n✅ Series embeddings complete\")\nprint(\"Shape:\", series_embeddings.shape)\nprint(\"Min:\", series_embeddings.min())\nprint(\"Max:\", series_embeddings.max())\nprint(\"Mean:\", series_embeddings.mean())\nprint(\"Std:\", series_embeddings.std())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T04:05:49.102033Z","iopub.execute_input":"2026-08-12T04:05:49.102804Z","iopub.status.idle":"2026-08-12T04:07:01.708806Z","shell.execute_reply.started":"2026-08-12T04:05:49.10273Z","shell.execute_reply":"2026-08-12T04:07:01.707955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\n\nwith open(\"/kaggle/working/series_embeddings_58.pkl\", \"wb\") as f:\n    pickle.dump(series_embeddings, f)\n\nprint(\"✅ Series embeddings saved!\")\nprint(\"Shape:\", series_embeddings.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T04:07:52.760845Z","iopub.execute_input":"2026-08-12T04:07:52.761509Z","iopub.status.idle":"2026-08-12T04:07:52.769967Z","shell.execute_reply.started":"2026-08-12T04:07:52.761471Z","shell.execute_reply":"2026-08-12T04:07:52.768912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Mean pooling across the 5 series\nX_mean = series_embeddings.mean(axis=1)\n\n# Max pooling across the 5 series\nX_max = series_embeddings.max(axis=1)\n\n# Our previous champion: 0.7 Mean + 0.3 Max\nX_weighted = 0.7 * X_mean + 0.3 * X_max\n\nprint(\"✅ Pooling representations created\")\n\nprint(\"Mean:\", X_mean.shape)\nprint(\"Max :\", X_max.shape)\nprint(\"Weighted:\", X_weighted.shape)\n\nprint(\"\\nWeighted statistics:\")\nprint(\"Mean:\", X_weighted.mean())\nprint(\"Std :\", X_weighted.std())\nprint(\"Min :\", X_weighted.min())\nprint(\"Max :\", X_weighted.max())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T04:08:36.923881Z","iopub.execute_input":"2026-08-12T04:08:36.924878Z","iopub.status.idle":"2026-08-12T04:08:36.934905Z","shell.execute_reply.started":"2026-08-12T04:08:36.924672Z","shell.execute_reply":"2026-08-12T04:08:36.933754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import LeaveOneOut\n\n# Make sure labels are available\nY = labeled_train[label_cols].values.astype(float)\n\ndef evaluate_loo(X, Y):\n    loo = LeaveOneOut()\n\n    predictions = np.zeros_like(Y, dtype=float)\n\n    for train_idx, test_idx in loo.split(X):\n\n        model = make_pipeline(\n            StandardScaler(),\n            LogisticRegression(\n                max_iter=3000,\n                C=1.0,\n                solver=\"liblinear\"\n            )\n        )\n\n        # Fit one label at a time\n        for j in range(Y.shape[1]):\n\n            model.fit(\n                X[train_idx],\n                Y[train_idx, j]\n            )\n\n            predictions[test_idx, j] = model.predict_proba(\n                X[test_idx]\n            )[:, 1][0]\n\n    aucs = {}\n\n    for j, col in enumerate(label_cols):\n        aucs[col] = roc_auc_score(\n            Y[:, j],\n            predictions[:, j]\n        )\n\n    macro_auc = np.mean(list(aucs.values()))\n\n    return aucs, macro_auc, predictions\n\n\n# ------------------------------------------------------------\n# Evaluate current champion\n# ------------------------------------------------------------\n\naucs, macro_auc, weighted_predictions = evaluate_loo(\n    X_weighted,\n    Y\n)\n\nprint(\"=\" * 50)\nprint(\"0.7 MEAN + 0.3 MAX\")\nprint(\"=\" * 50)\n\nfor name, auc in aucs.items():\n    print(f\"{name:20s} {auc:.4f}\")\n\nprint(f\"\\nMacro AUC: {macro_auc:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T04:09:32.089275Z","iopub.execute_input":"2026-08-12T04:09:32.090573Z","iopub.status.idle":"2026-08-12T04:09:41.800305Z","shell.execute_reply.started":"2026-08-12T04:09:32.090505Z","shell.execute_reply":"2026-08-12T04:09:41.798993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pickle\n\nbackup = {}\n\nfor name in [\n    \"all_mri_data\",\n    \"X_mri\",\n    \"series_embeddings\",\n    \"X_mean\",\n    \"X_max\",\n    \"X_weighted\",\n    \"Y\",\n    \"Y_mri\",\n    \"label_cols\",\n    \"weighted_predictions\"\n]:\n    if name in globals():\n        backup[name] = globals()[name]\n\nbackup_path = \"/kaggle/working/MRI_MASTER_BACKUP.pkl\"\n\nwith open(backup_path, \"wb\") as f:\n    pickle.dump(backup, f)\n\nprint(\"✅ MASTER BACKUP SAVED\")\nprint(\"File:\", backup_path)\nprint(\"Saved variables:\", list(backup.keys()))\nprint(\"File exists:\", os.path.exists(backup_path))\nprint(\"File size MB:\", round(os.path.getsize(backup_path) / 1024**2, 2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T04:12:36.720619Z","iopub.execute_input":"2026-08-12T04:12:36.721039Z","iopub.status.idle":"2026-08-12T04:12:38.429532Z","shell.execute_reply.started":"2026-08-12T04:12:36.720997Z","shell.execute_reply":"2026-08-12T04:12:38.427644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"=== CURRENT CHECKPOINT ===\")\nprint(\"all_mri_data:\", all_mri_data.shape if hasattr(all_mri_data, \"shape\") else type(all_mri_data))\nprint(\"X_mri:\", X_mri.shape)\nprint(\"series_embeddings:\", series_embeddings.shape)\nprint(\"X_mean:\", X_mean.shape)\nprint(\"X_max:\", X_max.shape)\nprint(\"X_weighted:\", X_weighted.shape)\nprint(\"Y:\", Y.shape)\nprint(\"label_cols:\", len(label_cols))\n\nprint(\"\\n=== BACKUP ===\")\nimport os\nbackup_path = \"/kaggle/working/MRI_MASTER_BACKUP.pkl\"\nprint(\"Backup exists:\", os.path.exists(backup_path))\nprint(\"Backup size MB:\", round(os.path.getsize(backup_path) / 1024**2, 2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T04:13:17.773033Z","iopub.execute_input":"2026-08-12T04:13:17.773463Z","iopub.status.idle":"2026-08-12T04:13:17.782485Z","shell.execute_reply.started":"2026-08-12T04:13:17.77343Z","shell.execute_reply":"2026-08-12T04:13:17.781283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"=== LABEL CHECK ===\")\nprint(\"Y shape:\", Y.shape)\nprint(\"Labels:\", label_cols)\n\nprint(\"\\nPositive counts:\")\nfor i, col in enumerate(label_cols):\n    print(f\"{col:20s} {int(Y[:, i].sum())}\")\n\nprint(\"\\n=== WEIGHTED EMBEDDING CHECK ===\")\nprint(\"Shape:\", X_weighted.shape)\nprint(\"Mean:\", X_weighted.mean())\nprint(\"Std :\", X_weighted.std())\nprint(\"Min :\", X_weighted.min())\nprint(\"Max :\", X_weighted.max())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T04:14:00.111187Z","iopub.execute_input":"2026-08-12T04:14:00.111729Z","iopub.status.idle":"2026-08-12T04:14:00.12218Z","shell.execute_reply.started":"2026-08-12T04:14:00.111685Z","shell.execute_reply":"2026-08-12T04:14:00.120698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import LeaveOneOut\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import roc_auc_score\nimport numpy as np\n\ndef evaluate_C(X, Y, C):\n    loo = LeaveOneOut()\n    predictions = np.zeros_like(Y, dtype=float)\n\n    for train_idx, test_idx in loo.split(X):\n\n        for j in range(Y.shape[1]):\n\n            model = make_pipeline(\n                StandardScaler(),\n                LogisticRegression(\n                    max_iter=3000,\n                    C=C,\n                    solver=\"liblinear\"\n                )\n            )\n\n            model.fit(X[train_idx], Y[train_idx, j])\n\n            predictions[test_idx, j] = (\n                model.predict_proba(X[test_idx])[:, 1][0]\n            )\n\n    aucs = [\n        roc_auc_score(Y[:, j], predictions[:, j])\n        for j in range(Y.shape[1])\n    ]\n\n    return np.mean(aucs)\n\n\nprint(\"=== C VALUE CHECK ===\")\n\nfor C in [0.01, 0.03, 0.1, 0.3, 1.0, 3.0, 10.0]:\n    score = evaluate_C(X_weighted, Y, C)\n    print(f\"C={C:<5}  Macro AUC={score:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T04:14:48.842743Z","iopub.execute_input":"2026-08-12T04:14:48.843351Z","iopub.status.idle":"2026-08-12T04:15:32.034392Z","shell.execute_reply.started":"2026-08-12T04:14:48.843318Z","shell.execute_reply":"2026-08-12T04:15:32.032951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import inspect\n\nfor name in [\n    \"load_series_images\",\n    \"build_study_tensor\",\n    \"normalize_mri\",\n    \"resize_mri\"\n]:\n    print(\"\\n\" + \"=\" * 70)\n    print(name)\n\n    if name in globals():\n        try:\n            print(inspect.getsource(globals()[name]))\n        except Exception as e:\n            print(\"Could not retrieve source:\", e)\n    else:\n        print(\"NOT AVAILABLE\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T04:17:28.980946Z","iopub.execute_input":"2026-08-12T04:17:28.981493Z","iopub.status.idle":"2026-08-12T04:17:28.992519Z","shell.execute_reply.started":"2026-08-12T04:17:28.981457Z","shell.execute_reply":"2026-08-12T04:17:28.991131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import inspect\n\nprint(\"=\" * 70)\nprint(\"extract_study_embedding\")\nprint(\"=\" * 70)\n\nif \"extract_study_embedding\" in globals():\n    print(inspect.getsource(extract_study_embedding))\nelse:\n    print(\"extract_study_embedding NOT AVAILABLE\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T04:19:35.239857Z","iopub.execute_input":"2026-08-12T04:19:35.240257Z","iopub.status.idle":"2026-08-12T04:19:35.248083Z","shell.execute_reply.started":"2026-08-12T04:19:35.240215Z","shell.execute_reply":"2026-08-12T04:19:35.246651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Studies:\", len(all_mri_data))\n\nfor i, (study_id, series_list) in enumerate(all_mri_data.items()):\n\n    actual = len(series_list)\n    used = X_mri[i].shape[0]\n\n    if actual != used:\n        print(\n            f\"Study {i}: actual series={actual}, \"\n            f\"X_mri series={used}\"\n        )\n\nprint(\"\\nDone.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T04:20:25.652044Z","iopub.execute_input":"2026-08-12T04:20:25.652639Z","iopub.status.idle":"2026-08-12T04:20:25.661386Z","shell.execute_reply.started":"2026-08-12T04:20:25.652606Z","shell.execute_reply":"2026-08-12T04:20:25.659889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"=== CURRENT all_mri_data CHECK ===\")\n\ncounts = [len(v) for v in all_mri_data.values()]\n\nprint(\"Number of studies:\", len(counts))\nprint(\"Unique series counts in all_mri_data:\", sorted(set(counts)))\nprint(\"Counts:\", counts)\n\nprint(\"\\nOriginal dataset counts:\")\nprint([len(v) for v in series_embeddings])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T04:21:32.485672Z","iopub.execute_input":"2026-08-12T04:21:32.486634Z","iopub.status.idle":"2026-08-12T04:21:32.495954Z","shell.execute_reply.started":"2026-08-12T04:21:32.48659Z","shell.execute_reply":"2026-08-12T04:21:32.493936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\nTRAIN_SERIES = \"/kaggle/input/competitions/rsna-knee-abnormality-detection/train_series/train_series.csv\"\n\n# Use the already-loaded train_series if available\nif \"train_series\" in globals():\n    ts = train_series.copy()\nelse:\n    ts = pd.read_csv(TRAIN_SERIES)\n\nprint(\"train_series shape:\", ts.shape)\nprint(\"Columns:\", list(ts.columns))\n\n# First labeled study\nstudy_id = list(all_mri_data.keys())[0]\n\nprint(\"\\nFirst labeled study:\")\nprint(study_id)\n\nrows = ts[ts[\"StudyInstanceUID\"] == study_id].copy()\n\nprint(\"\\nALL SERIES FOR THIS STUDY:\")\nprint(\n    rows[\n        [\n            \"SeriesInstanceUID\",\n            \"Fluid_Sensitive\",\n            \"Fat_Suppression\",\n            \"Anatomical_Plane\"\n        ]\n    ].to_string(index=False)\n)\n\nprint(\"\\nNumber of series in CSV:\", len(rows))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-12T04:22:38.203068Z","iopub.execute_input":"2026-08-12T04:22:38.204035Z","iopub.status.idle":"2026-08-12T04:22:38.252488Z","shell.execute_reply.started":"2026-08-12T04:22:38.203999Z","shell.execute_reply":"2026-08-12T04:22:38.251374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"=== RESNET CONFIGURATION ===\")\n\nprint(\"Model type:\", type(resnet))\n\nprint(\"\\nTraining mode:\", resnet.training)\n\nprint(\"\\nFinal layer:\")\nprint(resnet.fc)\n\nprint(\"\\nFirst convolution:\")\nprint(resnet.conv1)\n\nprint(\"\\nImageNet mean:\", mean)\nprint(\"ImageNet std :\", std)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-14T04:30:19.330687Z","iopub.execute_input":"2026-08-14T04:30:19.33086Z","iopub.status.idle":"2026-08-14T04:30:19.342574Z","shell.execute_reply.started":"2026-08-14T04:30:19.330837Z","shell.execute_reply":"2026-08-14T04:30:19.341725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"=== SAVED EMBEDDING CHECK ===\")\n\nprint(\"series_embeddings:\", series_embeddings.shape)\nprint(\"dtype:\", series_embeddings.dtype)\nprint(\"mean:\", series_embeddings.mean())\nprint(\"std :\", series_embeddings.std())\nprint(\"min :\", series_embeddings.min())\nprint(\"max :\", series_embeddings.max())\n\nprint(\"\\n=== CURRENT POOLING ===\")\n\nX_mean_check = series_embeddings.mean(axis=1)\nX_max_check = series_embeddings.max(axis=1)\nX_weighted_check = 0.7 * X_mean_check + 0.3 * X_max_check\n\nprint(\"Mean:\", X_mean_check.shape)\nprint(\"Max :\", X_max_check.shape)\nprint(\"Weighted:\", X_weighted_check.shape)\n\nprint(\"\\nDifference from saved X_mean:\",\n      np.max(np.abs(X_mean_check - X_mean)))\n\nprint(\"Difference from saved X_max:\",\n      np.max(np.abs(X_max_check - X_max)))\n\nprint(\"Difference from saved X_weighted:\",\n      np.max(np.abs(X_weighted_check - X_weighted)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-14T04:31:18.289213Z","iopub.execute_input":"2026-08-14T04:31:18.289431Z","iopub.status.idle":"2026-08-14T04:31:18.297232Z","shell.execute_reply.started":"2026-08-14T04:31:18.289411Z","shell.execute_reply":"2026-08-14T04:31:18.296473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\nimport os\n\nBACKUP = \"/kaggle/working/MRI_MASTER_BACKUP.pkl\"\n\nprint(\"Backup exists:\", os.path.exists(BACKUP))\nprint(\"Backup size MB:\", round(os.path.getsize(BACKUP) / 1024**2, 2))\n\nwith open(BACKUP, \"rb\") as f:\n    backup = pickle.load(f)\n\nprint(\"\\nBackup loaded safely.\")\nprint(\"Variables:\", list(backup.keys()))\n\n# Restore the saved objects\nall_mri_data = backup[\"all_mri_data\"]\nX_mri = backup[\"X_mri\"]\nseries_embeddings = backup[\"series_embeddings\"]\nX_mean = backup[\"X_mean\"]\nX_max = backup[\"X_max\"]\nX_weighted = backup[\"X_weighted\"]\nY = backup[\"Y\"]\nlabel_cols = backup[\"label_cols\"]\nweighted_predictions = backup[\"weighted_predictions\"]\n\nprint(\"\\n=== RESTORED ===\")\nprint(\"X_mri:\", X_mri.shape)\nprint(\"series_embeddings:\", series_embeddings.shape)\nprint(\"X_mean:\", X_mean.shape)\nprint(\"X_max:\", X_max.shape)\nprint(\"X_weighted:\", X_weighted.shape)\nprint(\"Y:\", Y.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-14T04:32:28.512763Z","iopub.execute_input":"2026-08-14T04:32:28.513005Z","iopub.status.idle":"2026-08-14T04:32:28.570286Z","shell.execute_reply.started":"2026-08-14T04:32:28.512979Z","shell.execute_reply":"2026-08-14T04:32:28.569502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\nimport pickle\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport cv2\n\n# ============================================================\n# RECOVER 58 LABELED STUDIES + MRI TENSOR\n# ============================================================\n\nBASE = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\n\nTRAIN = os.path.join(BASE, \"train.csv\")\nTRAIN_SERIES = os.path.join(BASE, \"train_series.csv\")\nSERIES_DIR = os.path.join(BASE, \"train_series\")\n\ntrain = pd.read_csv(TRAIN)\ntrain_series = pd.read_csv(TRAIN_SERIES)\n\nlabel_cols = [\n    \"ACL\",\n    \"MCL\",\n    \"Medial Meniscus\",\n    \"Lateral Meniscus\",\n    \"Medial OA\",\n    \"Lateral OA\",\n    \"PF OA\",\n    \"Effusion\",\n    \"Synovitis\",\n    \"Baker's\",\n    \"Contusion\",\n    \"Fracture\"\n]\n\n# ------------------------------------------------------------\n# 1. Select the same 58 labeled studies\n# ------------------------------------------------------------\n\nlabeled_train = train[\n    train[label_cols].notna().all(axis=1)\n].copy()\n\n# If the dataset contains exactly the intended 58, use them.\n# Otherwise take the first 58 fully labeled studies.\nif len(labeled_train) != 58:\n    labeled_train = labeled_train.iloc[:58].copy()\n\nstudy_ids = labeled_train[\"StudyInstanceUID\"].tolist()\n\nY = labeled_train[label_cols].values.astype(np.float32)\n\nprint(\"Labeled studies:\", len(study_ids))\nprint(\"Y shape:\", Y.shape)\n\n# ------------------------------------------------------------\n# 2. MRI preprocessing\n# ------------------------------------------------------------\n\ndef normalize_mri(img):\n    img = img.astype(np.float32)\n\n    lo = np.percentile(img, 1)\n    hi = np.percentile(img, 99)\n\n    if hi <= lo:\n        return np.zeros_like(img, dtype=np.float32)\n\n    img = np.clip(img, lo, hi)\n    img = (img - lo) / (hi - lo)\n\n    return img.astype(np.float32)\n\n\ndef resize_mri(img, size=224):\n    return cv2.resize(\n        img,\n        (size, size),\n        interpolation=cv2.INTER_AREA\n    ).astype(np.float32)\n\n\ndef load_series_images(series_path, n_slices=5):\n\n    records = []\n\n    for fname in os.listdir(series_path):\n\n        fpath = os.path.join(series_path, fname)\n\n        if not os.path.isfile(fpath):\n            continue\n\n        try:\n            ds = pydicom.dcmread(fpath)\n\n            pixel = ds.pixel_array.astype(np.float32)\n\n            instance = getattr(ds, \"InstanceNumber\", None)\n\n            if instance is None:\n                instance = len(records)\n\n            records.append((int(instance), pixel))\n\n        except Exception:\n            continue\n\n    if len(records) == 0:\n        return None\n\n    records.sort(key=lambda x: x[0])\n\n    indices = np.linspace(\n        0,\n        len(records) - 1,\n        n_slices,\n        dtype=int\n    )\n\n    selected = []\n\n    for idx in indices:\n\n        img = records[idx][1]\n        img = normalize_mri(img)\n        img = resize_mri(img, 224)\n\n        selected.append(img)\n\n    return np.stack(selected).astype(np.float32)\n\n\n# ------------------------------------------------------------\n# 3. Use ONLY the series listed in train_series.csv\n# ------------------------------------------------------------\n\nall_mri_data = {}\n\nfor n, study_id in enumerate(study_ids):\n\n    rows = train_series[\n        train_series[\"StudyInstanceUID\"] == study_id\n    ]\n\n    series_list = []\n\n    for _, row in rows.iterrows():\n\n        series_id = row[\"SeriesInstanceUID\"]\n\n        series_path = os.path.join(\n            SERIES_DIR,\n            study_id,\n            series_id\n        )\n\n        if not os.path.isdir(series_path):\n            continue\n\n        arr = load_series_images(\n            series_path,\n            n_slices=5\n        )\n\n        if arr is not None:\n            series_list.append(arr)\n\n    if len(series_list) != 5:\n        print(\n            \"WARNING:\",\n            study_id,\n            \"usable series:\",\n            len(series_list)\n        )\n\n    all_mri_data[study_id] = series_list\n\n    if (n + 1) % 10 == 0:\n        print(f\"Processed {n + 1}/58 studies\")\n\n\n# ------------------------------------------------------------\n# 4. Build tensor\n# ------------------------------------------------------------\n\nX_mri = np.stack(\n    [\n        np.stack(all_mri_data[sid])\n        for sid in study_ids\n    ]\n).astype(np.float32)\n\nprint(\"\\n========================================\")\nprint(\"MRI RECOVERY COMPLETE\")\nprint(\"========================================\")\nprint(\"Studies:\", len(study_ids))\nprint(\"X_mri:\", X_mri.shape)\nprint(\"Y:\", Y.shape)\nprint(\"Min:\", X_mri.min())\nprint(\"Max:\", X_mri.max())\nprint(\"Mean:\", X_mri.mean())\nprint(\"Std:\", X_mri.std())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-14T04:34:00.323221Z","iopub.execute_input":"2026-08-14T04:34:00.323493Z","iopub.status.idle":"2026-08-14T04:35:51.829887Z","shell.execute_reply.started":"2026-08-14T04:34:00.323462Z","shell.execute_reply":"2026-08-14T04:35:51.829104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\nimport os\n\npath = \"/kaggle/working/all_mri_data_variable.pkl\"\n\nwith open(path, \"wb\") as f:\n    pickle.dump(\n        {\n            \"all_mri_data\": all_mri_data,\n            \"study_ids\": study_ids,\n            \"Y\": Y,\n            \"label_cols\": label_cols,\n        },\n        f,\n        protocol=pickle.HIGHEST_PROTOCOL\n    )\n\nprint(\"✅ VARIABLE-SERIES MRI DATA SAVED\")\nprint(\"File:\", path)\nprint(\"Size MB:\", round(os.path.getsize(path) / 1024**2, 2))\nprint(\"Studies:\", len(all_mri_data))\nprint(\"Series counts:\", [len(v) for v in all_mri_data.values()])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-14T04:39:16.776613Z","iopub.execute_input":"2026-08-14T04:39:16.77683Z","iopub.status.idle":"2026-08-14T04:39:16.982812Z","shell.execute_reply.started":"2026-08-14T04:39:16.776811Z","shell.execute_reply":"2026-08-14T04:39:16.982295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torchvision.models as models\nimport numpy as np\n\nprint(\"Loading ResNet18...\")\n\nweights = models.ResNet18_Weights.DEFAULT\nresnet = models.resnet18(weights=weights)\n\n# Remove classification layer\nresnet.fc = torch.nn.Identity()\n\nresnet.eval()\n\ndevice = torch.device(\"cpu\")\nresnet = resnet.to(device)\n\n# ImageNet normalization\nmean = torch.tensor(\n    [0.485, 0.456, 0.406],\n    dtype=torch.float32\n).view(1, 3, 1, 1)\n\nstd = torch.tensor(\n    [0.229, 0.224, 0.225],\n    dtype=torch.float32\n).view(1, 3, 1, 1)\n\n# Test first study\nfirst_study = study_ids[0]\nfirst_series = all_mri_data[first_study][0]\n\nimages = torch.from_numpy(first_series).float()\n\n# (5,224,224) -> (5,3,224,224)\nimages = images.unsqueeze(1).repeat(1, 3, 1, 1)\n\nimages = (images - mean) / std\n\nwith torch.no_grad():\n    emb = resnet(images)\n\nprint(\"\\n=== TEST RESULT ===\")\nprint(\"Input:\", images.shape)\nprint(\"Embedding:\", emb.shape)\nprint(\"Embedding mean:\", emb.mean().item())\nprint(\"Embedding std :\", emb.std().item())\nprint(\"Embedding min :\", emb.min().item())\nprint(\"Embedding max :\", emb.max().item())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-14T04:39:53.307777Z","iopub.execute_input":"2026-08-14T04:39:53.308004Z","iopub.status.idle":"2026-08-14T04:40:00.80787Z","shell.execute_reply.started":"2026-08-14T04:39:53.307986Z","shell.execute_reply":"2026-08-14T04:40:00.80719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pickle\nimport numpy as np\nimport torch\nfrom tqdm.auto import tqdm\n\nEMB_FILE = \"/kaggle/working/variable_series_embeddings.pkl\"\n\n# ------------------------------------------------------------\n# Resume if partial file already exists\n# ------------------------------------------------------------\n\nif os.path.exists(EMB_FILE):\n\n    with open(EMB_FILE, \"rb\") as f:\n        saved = pickle.load(f)\n\n    study_embeddings_variable = saved[\"study_embeddings\"]\n    completed_ids = saved[\"completed_ids\"]\n\n    print(\"Existing checkpoint found.\")\n    print(\"Completed:\", len(completed_ids))\n\nelse:\n\n    study_embeddings_variable = {}\n    completed_ids = []\n\n# ------------------------------------------------------------\n# Extract study embeddings\n# ------------------------------------------------------------\n\nresnet.eval()\n\nfor study_id in tqdm(\n    study_ids,\n    desc=\"Embedding studies\"\n):\n\n    if study_id in study_embeddings_variable:\n        continue\n\n    series_embeddings_this_study = []\n\n    for series in all_mri_data[study_id]:\n\n        images = torch.from_numpy(\n            np.asarray(series)\n        ).float()\n\n        # (5,224,224)\n        # -> (5,3,224,224)\n        images = images.unsqueeze(1).repeat(\n            1, 3, 1, 1\n        )\n\n        # ImageNet normalization\n        images = (\n            images - mean\n        ) / std\n\n        with torch.no_grad():\n\n            embeddings = resnet(\n                images\n            )\n\n        # Average 5 slices\n        series_embedding = (\n            embeddings.mean(dim=0)\n            .cpu()\n            .numpy()\n        )\n\n        series_embeddings_this_study.append(\n            series_embedding\n        )\n\n    # Average ALL available series\n    study_embedding = np.mean(\n        np.stack(series_embeddings_this_study),\n        axis=0\n    ).astype(np.float32)\n\n    study_embeddings_variable[study_id] = study_embedding\n\n    completed_ids.append(study_id)\n\n    # --------------------------------------------------------\n    # SAVE AFTER EVERY STUDY\n    # --------------------------------------------------------\n\n    with open(EMB_FILE, \"wb\") as f:\n\n        pickle.dump(\n            {\n                \"study_embeddings\":\n                    study_embeddings_variable,\n                \"completed_ids\":\n                    completed_ids\n            },\n            f,\n            protocol=pickle.HIGHEST_PROTOCOL\n        )\n\nprint(\"\\n========================================\")\nprint(\"VARIABLE-SERIES EMBEDDINGS COMPLETE\")\nprint(\"========================================\")\n\nprint(\n    \"Studies:\",\n    len(study_embeddings_variable)\n)\n\nprint(\n    \"Embedding dimension:\",\n    next(iter(\n        study_embeddings_variable.values()\n    )).shape\n)\n\nX_variable = np.stack([\n    study_embeddings_variable[sid]\n    for sid in study_ids\n]).astype(np.float32)\n\nprint(\"X_variable:\", X_variable.shape)\n\nprint(\"Mean:\", X_variable.mean())\nprint(\"Std :\", X_variable.std())\nprint(\"Min :\", X_variable.min())\nprint(\"Max :\", X_variable.max())\n\nprint(\"\\nSaved:\", EMB_FILE)\nprint(\n    \"Size MB:\",\n    round(\n        os.path.getsize(EMB_FILE) / 1024**2,\n        2\n    )\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-14T04:41:07.992591Z","iopub.execute_input":"2026-08-14T04:41:07.992844Z","iopub.status.idle":"2026-08-14T04:41:59.705384Z","shell.execute_reply.started":"2026-08-14T04:41:07.992823Z","shell.execute_reply":"2026-08-14T04:41:59.704895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import LeaveOneOut\nfrom sklearn.metrics import roc_auc_score\nimport numpy as np\n\nprint(\"=== VARIABLE-SERIES EMBEDDING EVALUATION ===\")\n\nX = X_variable\nY_eval = Y\n\nloo = LeaveOneOut()\n\npredictions = np.zeros(\n    (len(Y_eval), Y_eval.shape[1]),\n    dtype=np.float32\n)\n\naucs = []\n\nfor label_idx, label in enumerate(label_cols):\n\n    y = Y_eval[:, label_idx]\n\n    pred = np.zeros(len(y))\n\n    for train_idx, test_idx in loo.split(X):\n\n        model = make_pipeline(\n            StandardScaler(),\n            LogisticRegression(\n                C=0.01,\n                max_iter=5000,\n                class_weight=\"balanced\"\n            )\n        )\n\n        model.fit(\n            X[train_idx],\n            y[train_idx]\n        )\n\n        pred[test_idx] = model.predict_proba(\n            X[test_idx]\n        )[0, 1]\n\n    predictions[:, label_idx] = pred\n\n    # AUC is undefined if a fold/test setup has only one class,\n    # but with LOO we calculate it over all predictions.\n    auc = roc_auc_score(y, pred)\n\n    aucs.append(auc)\n\n    print(\n        f\"{label:18s} {auc:.4f}\"\n    )\n\nmacro_auc = np.mean(aucs)\n\nprint(\"\\n\" + \"=\" * 50)\nprint(f\"Macro AUC: {macro_auc:.4f}\")\nprint(\"=\" * 50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-14T04:43:03.918341Z","iopub.execute_input":"2026-08-14T04:43:03.918992Z","iopub.status.idle":"2026-08-14T04:43:09.135537Z","shell.execute_reply.started":"2026-08-14T04:43:03.918968Z","shell.execute_reply":"2026-08-14T04:43:09.135101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\nimport os\n\nCHAMPION_FILE = \"/kaggle/working/CHAMPION_06582.pkl\"\n\nchampion = {\n    \"X_mean\": X_mean,\n    \"X_max\": X_max,\n    \"X_weighted\": X_weighted,\n    \"Y\": Y,\n    \"label_cols\": label_cols,\n    \"macro_auc\": 0.6582475022187056,\n    \"mean_weight\": 0.7,\n    \"max_weight\": 0.3\n}\n\nwith open(CHAMPION_FILE, \"wb\") as f:\n    pickle.dump(\n        champion,\n        f,\n        protocol=pickle.HIGHEST_PROTOCOL\n    )\n\nprint(\"✅ CHAMPION CHECKPOINT SAVED\")\nprint(\"File:\", CHAMPION_FILE)\nprint(\n    \"Size MB:\",\n    round(os.path.getsize(CHAMPION_FILE) / 1024**2, 2)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-14T04:44:25.217625Z","iopub.execute_input":"2026-08-14T04:44:25.217868Z","iopub.status.idle":"2026-08-14T04:44:25.223756Z","shell.execute_reply.started":"2026-08-14T04:44:25.217847Z","shell.execute_reply":"2026-08-14T04:44:25.223174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluate_embedding_loo(X, Y):\n\n    predictions = np.zeros_like(Y, dtype=float)\n\n    for j, label in enumerate(label_cols):\n\n        y = Y[:, j]\n\n        for train_idx, test_idx in LeaveOneOut().split(X):\n\n            model = make_pipeline(\n                StandardScaler(),\n                LogisticRegression(\n                    C=1.0,\n                    max_iter=3000,\n                    class_weight=\"balanced\"\n                )\n            )\n\n            # One target only\n            model.fit(\n                X[train_idx],\n                y[train_idx]\n            )\n\n            predictions[test_idx, j] = model.predict_proba(\n                X[test_idx]\n            )[0, 1]\n\n    aucs = {}\n\n    for j, label in enumerate(label_cols):\n        aucs[label] = roc_auc_score(\n            Y[:, j],\n            predictions[:, j]\n        )\n\n    macro = np.mean(list(aucs.values()))\n\n    return aucs, macro, predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T10:11:31.010093Z","iopub.execute_input":"2026-08-09T10:11:31.01046Z","iopub.status.idle":"2026-08-09T10:11:31.021469Z","shell.execute_reply.started":"2026-08-09T10:11:31.010428Z","shell.execute_reply":"2026-08-09T10:11:31.020178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import roc_auc_score\nimport numpy as np\n\nY_mri = report_label_df[target_cols].astype(int).values\n\nmri_pred = np.zeros_like(Y_mri, dtype=float)\n\nfor i in range(len(X_mri_features)):\n\n    train_idx = np.arange(len(X_mri_features)) != i\n\n    model = make_pipeline(\n        StandardScaler(),\n        LogisticRegression(\n            max_iter=3000,\n            class_weight=\"balanced\"\n        )\n    )\n\n    model.fit(\n        X_mri_features[train_idx],\n        Y_mri[train_idx]\n    )\n\n    mri_pred[i] = model.predict_proba(\n        X_mri_features[i:i+1]\n    )[:, 1]\n\nprint(\"MRI predictions:\", mri_pred.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:22:28.388971Z","iopub.execute_input":"2026-08-09T08:22:28.389308Z","iopub.status.idle":"2026-08-09T08:22:28.416005Z","shell.execute_reply.started":"2026-08-09T08:22:28.389281Z","shell.execute_reply":"2026-08-09T08:22:28.414421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mri_auc = {}\n\nfor j, target in enumerate(target_cols):\n    mri_auc[target] = roc_auc_score(\n        Y_mri[:, j],\n        mri_pred[:, j]\n    )\n\nfor target, score in mri_auc.items():\n    print(f\"{target:20s} {score:.4f}\")\n\nprint(\"\\nMRI Macro AUC:\",\n      roc_auc_score(Y_mri, mri_pred, average=\"macro\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:22:45.166384Z","iopub.execute_input":"2026-08-09T08:22:45.166804Z","iopub.status.idle":"2026-08-09T08:22:45.208978Z","shell.execute_reply.started":"2026-08-09T08:22:45.166772Z","shell.execute_reply":"2026-08-09T08:22:45.207911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\n\nprint(\"PyTorch:\", torch.__version__)\nprint(\"CUDA available:\", torch.cuda.is_available())\n\nif torch.cuda.is_available():\n    print(\"GPU:\", torch.cuda.get_device_name(0))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:26:20.206014Z","iopub.execute_input":"2026-08-09T08:26:20.206484Z","iopub.status.idle":"2026-08-09T08:26:25.237461Z","shell.execute_reply.started":"2026-08-09T08:26:20.206414Z","shell.execute_reply":"2026-08-09T08:26:25.235724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import importlib.util\n\nprint(\"torchvision:\", importlib.util.find_spec(\"torchvision\"))\nprint(\"timm:\", importlib.util.find_spec(\"timm\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:27:04.789735Z","iopub.execute_input":"2026-08-09T08:27:04.790092Z","iopub.status.idle":"2026-08-09T08:27:04.800695Z","shell.execute_reply.started":"2026-08-09T08:27:04.790062Z","shell.execute_reply":"2026-08-09T08:27:04.799652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torchvision.models as models\n\ndevice = torch.device(\"cpu\")\n\nweights = models.ResNet18_Weights.DEFAULT\n\nresnet = models.resnet18(weights=weights)\n\n# Remove the final classification layer\nresnet.fc = torch.nn.Identity()\n\nresnet = resnet.to(device)\nresnet.eval()\n\n# Freeze everything\nfor param in resnet.parameters():\n    param.requires_grad = False\n\nprint(\"Model loaded.\")\nprint(\"Parameters:\", sum(p.numel() for p in resnet.parameters()))\nprint(\"Trainable:\", sum(p.numel() for p in resnet.parameters() if p.requires_grad))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:27:55.562564Z","iopub.execute_input":"2026-08-09T08:27:55.562978Z","iopub.status.idle":"2026-08-09T08:28:04.113946Z","shell.execute_reply.started":"2026-08-09T08:27:55.562947Z","shell.execute_reply":"2026-08-09T08:28:04.112687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport numpy as np\nfrom tqdm.auto import tqdm\n\n# ImageNet normalization expected by ResNet18\nmean = torch.tensor(\n    [0.485, 0.456, 0.406]\n).view(3, 1, 1)\n\nstd = torch.tensor(\n    [0.229, 0.224, 0.225]\n).view(3, 1, 1)\n\n\ndef extract_study_embedding(series_list):\n    \"\"\"\n    Input:\n        list of series\n        each series = (5, 224, 224)\n\n    Output:\n        512-dimensional study embedding\n    \"\"\"\n\n    all_embeddings = []\n\n    with torch.no_grad():\n\n        for series in series_list:\n\n            # series: (5, 224, 224)\n            images = torch.from_numpy(\n                np.asarray(series)\n            ).float()\n\n            # (5, 224, 224) -> (5, 3, 224, 224)\n            images = images.unsqueeze(1).repeat(1, 3, 1, 1)\n\n            # ImageNet normalization\n            images = (images - mean) / std\n\n            embeddings = resnet(\n                images\n            )\n\n            # Average the 5 slices\n            series_embedding = embeddings.mean(dim=0)\n\n            all_embeddings.append(\n                series_embedding.numpy()\n            )\n\n    # Average all available series\n    study_embedding = np.mean(\n        np.stack(all_embeddings),\n        axis=0\n    )\n\n    return study_embedding.astype(np.float32)\n\n\n# --------------------------------------------------\n# Extract embeddings for all 58 studies\n# --------------------------------------------------\n\nX_mri_embed = []\n\nfor study_id in tqdm(\n    labeled_ids,\n    desc=\"Extracting MRI embeddings\"\n):\n\n    embedding = extract_study_embedding(\n        all_mri_data[study_id]\n    )\n\n    X_mri_embed.append(embedding)\n\n\nX_mri_embed = np.stack(X_mri_embed)\n\nprint(\"MRI embedding shape:\", X_mri_embed.shape)\nprint(\"Min:\", X_mri_embed.min())\nprint(\"Max:\", X_mri_embed.max())\nprint(\"Mean:\", X_mri_embed.mean())\nprint(\"Std:\", X_mri_embed.std())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:29:34.824029Z","iopub.execute_input":"2026-08-09T08:29:34.824419Z","iopub.status.idle":"2026-08-09T08:30:47.770992Z","shell.execute_reply.started":"2026-08-09T08:29:34.824387Z","shell.execute_reply":"2026-08-09T08:30:47.770016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import LeaveOneOut\nfrom sklearn.metrics import roc_auc_score\n\nX = X_mri_embed\nY = labels_df[label_cols].values.astype(float)\n\nloo = LeaveOneOut()\n\npredictions = np.zeros_like(Y, dtype=float)\n\nfor train_idx, val_idx in loo.split(X):\n\n    X_train = X[train_idx]\n    X_val = X[val_idx]\n\n    for j, target in enumerate(label_cols):\n\n        y_train = Y[train_idx, j]\n\n        # Both classes must exist in training data\n        if len(np.unique(y_train)) < 2:\n            predictions[val_idx[0], j] = y_train.mean()\n            continue\n\n        model = LogisticRegression(\n            C=0.1,\n            max_iter=1000,\n            class_weight=\"balanced\",\n            solver=\"liblinear\"\n        )\n\n        model.fit(X_train, y_train)\n\n        predictions[val_idx[0], j] = (\n            model.predict_proba(X_val)[0, 1]\n        )\n\n\nprint(\"Predictions shape:\", predictions.shape)\n\nauc_scores = {}\n\nfor j, target in enumerate(label_cols):\n\n    try:\n        auc = roc_auc_score(\n            Y[:, j],\n            predictions[:, j]\n        )\n    except ValueError:\n        auc = np.nan\n\n    auc_scores[target] = auc\n\nprint(\"\\nPer-target AUC:\")\n\nfor target, auc in auc_scores.items():\n    print(f\"{target:<20} {auc:.4f}\")\n\nmacro_auc = np.nanmean(\n    list(auc_scores.values())\n)\n\nprint(\"\\nMRI Embedding Macro AUC:\", macro_auc)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:31:55.141696Z","iopub.execute_input":"2026-08-09T08:31:55.142044Z","iopub.status.idle":"2026-08-09T08:31:55.156092Z","shell.execute_reply.started":"2026-08-09T08:31:55.142015Z","shell.execute_reply":"2026-08-09T08:31:55.154639Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"label_cols = [\n    \"ACL\",\n    \"MCL\",\n    \"Medial Meniscus\",\n    \"Lateral Meniscus\",\n    \"Medial OA\",\n    \"Lateral OA\",\n    \"PF OA\",\n    \"Effusion\",\n    \"Synovitis\",\n    \"Baker's\",\n    \"Contusion\",\n    \"Fracture\"\n]\n\n# Use the 58 labeled studies already prepared\nY = labeled_df[label_cols].values.astype(float)\n\nprint(\"X shape:\", X_mri_embed.shape)\nprint(\"Y shape:\", Y.shape)\nprint(\"Targets:\", label_cols)","metadata":{}},{"cell_type":"code","source":"label_cols = [\n    \"ACL\",\n    \"MCL\",\n    \"Medial Meniscus\",\n    \"Lateral Meniscus\",\n    \"Medial OA\",\n    \"Lateral OA\",\n    \"PF OA\",\n    \"Effusion\",\n    \"Synovitis\",\n    \"Baker's\",\n    \"Contusion\",\n    \"Fracture\"\n]\n\n# Use the 58 labeled studies already prepared\nY = labeled_df[label_cols].values.astype(float)\n\nprint(\"X shape:\", X_mri_embed.shape)\nprint(\"Y shape:\", Y.shape)\nprint(\"Targets:\", label_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:33:22.436894Z","iopub.execute_input":"2026-08-09T08:33:22.438404Z","iopub.status.idle":"2026-08-09T08:33:22.448992Z","shell.execute_reply.started":"2026-08-09T08:33:22.438359Z","shell.execute_reply":"2026-08-09T08:33:22.447248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nfor name, obj in globals().items():\n    if isinstance(obj, pd.DataFrame):\n        cols = set(obj.columns)\n        if \"StudyInstanceUID\" in cols:\n            print(\n                name,\n                \"shape =\", obj.shape,\n                \"label columns =\", len(cols.intersection(set([\n                    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\",\n                    \"Medial OA\", \"Lateral OA\", \"PF OA\", \"Effusion\",\n                    \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\"\n                ])))\n            )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:33:58.830761Z","iopub.execute_input":"2026-08-09T08:33:58.831139Z","iopub.status.idle":"2026-08-09T08:33:58.842952Z","shell.execute_reply.started":"2026-08-09T08:33:58.831109Z","shell.execute_reply":"2026-08-09T08:33:58.841425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Make a fixed copy of the current globals first\nall_vars = list(globals().items())\n\nfor name, obj in all_vars:\n    if isinstance(obj, pd.DataFrame):\n        cols = set(obj.columns)\n\n        label_count = len(cols.intersection({\n            \"ACL\",\n            \"MCL\",\n            \"Medial Meniscus\",\n            \"Lateral Meniscus\",\n            \"Medial OA\",\n            \"Lateral OA\",\n            \"PF OA\",\n            \"Effusion\",\n            \"Synovitis\",\n            \"Baker's\",\n            \"Contusion\",\n            \"Fracture\"\n        }))\n\n        if \"StudyInstanceUID\" in cols:\n            print(\n                name,\n                \"| shape:\", obj.shape,\n                \"| label columns:\", label_count\n            )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:34:37.405371Z","iopub.execute_input":"2026-08-09T08:34:37.405742Z","iopub.status.idle":"2026-08-09T08:34:37.415366Z","shell.execute_reply.started":"2026-08-09T08:34:37.405712Z","shell.execute_reply":"2026-08-09T08:34:37.414012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_cols = [\n    \"ACL\",\n    \"MCL\",\n    \"Medial Meniscus\",\n    \"Lateral Meniscus\",\n    \"Medial OA\",\n    \"Lateral OA\",\n    \"PF OA\",\n    \"Effusion\",\n    \"Synovitis\",\n    \"Baker's\",\n    \"Contusion\",\n    \"Fracture\"\n]\n\nX = X_mri_embed\nY = labeled_train[label_cols].values.astype(float)\n\nprint(\"MRI embeddings:\", X.shape)\nprint(\"Labels:\", Y.shape)\n\nprint(\"\\nStudy IDs:\")\nprint(\"MRI IDs:\", len(labeled_ids))\nprint(\"Label IDs:\", len(labeled_train))\n\n# Verify the ordering before doing any ML\nprint(\"\\nFirst MRI ID:\", labeled_ids[0])\nprint(\"First label ID:\", labeled_train[\"StudyInstanceUID\"].iloc[0])\n\nprint(\"\\nIDs aligned:\",\n      list(labeled_ids) == list(labeled_train[\"StudyInstanceUID\"]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:35:29.741861Z","iopub.execute_input":"2026-08-09T08:35:29.742206Z","iopub.status.idle":"2026-08-09T08:35:29.752187Z","shell.execute_reply.started":"2026-08-09T08:35:29.742176Z","shell.execute_reply":"2026-08-09T08:35:29.75103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import LeaveOneOut\nfrom sklearn.metrics import roc_auc_score\nimport numpy as np\n\nloo = LeaveOneOut()\n\nmri_predictions = np.zeros((58, 12), dtype=float)\n\nfor train_idx, val_idx in loo.split(X):\n\n    X_train = X[train_idx]\n    X_val = X[val_idx]\n\n    for j, target in enumerate(label_cols):\n\n        y_train = Y[train_idx, j]\n\n        if len(np.unique(y_train)) < 2:\n            mri_predictions[val_idx[0], j] = y_train.mean()\n            continue\n\n        model = LogisticRegression(\n            C=0.1,\n            class_weight=\"balanced\",\n            solver=\"liblinear\",\n            max_iter=1000\n        )\n\n        model.fit(X_train, y_train)\n\n        mri_predictions[val_idx[0], j] = \\\n            model.predict_proba(X_val)[0, 1]\n\n\nprint(\"Predictions shape:\", mri_predictions.shape)\n\nmri_auc = {}\n\nfor j, target in enumerate(label_cols):\n    mri_auc[target] = roc_auc_score(\n        Y[:, j],\n        mri_predictions[:, j]\n    )\n\nprint(\"\\nMRI ResNet embedding AUC:\")\n\nfor target, auc in mri_auc.items():\n    print(f\"{target:<20} {auc:.4f}\")\n\nmri_macro_auc = np.mean(\n    list(mri_auc.values())\n)\n\nprint(\"\\nMRI Macro AUC:\", mri_macro_auc)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:36:07.714671Z","iopub.execute_input":"2026-08-09T08:36:07.715075Z","iopub.status.idle":"2026-08-09T08:36:10.47709Z","shell.execute_reply.started":"2026-08-09T08:36:07.715043Z","shell.execute_reply":"2026-08-09T08:36:10.47594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"MRI predictions:\", mri_predictions.shape)\n\n# Find existing prediction arrays from the notebook\nimport numpy as np\n\nfor name, obj in list(globals().items()):\n    if isinstance(obj, np.ndarray):\n        if obj.shape == (58, 12):\n            print(name, obj.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:37:41.054904Z","iopub.execute_input":"2026-08-09T08:37:41.055291Z","iopub.status.idle":"2026-08-09T08:37:41.064461Z","shell.execute_reply.started":"2026-08-09T08:37:41.05526Z","shell.execute_reply":"2026-08-09T08:37:41.062786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\nimport numpy as np\n\nreport_auc = []\n\nfor j, target in enumerate(label_cols):\n    auc = roc_auc_score(Y[:, j], pred[:, j])\n    report_auc.append(auc)\n    print(f\"{target:<20} {auc:.4f}\")\n\nprint(\"\\nReport Macro AUC:\", np.mean(report_auc))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:38:17.395224Z","iopub.execute_input":"2026-08-09T08:38:17.395833Z","iopub.status.idle":"2026-08-09T08:38:17.438857Z","shell.execute_reply.started":"2026-08-09T08:38:17.395783Z","shell.execute_reply":"2026-08-09T08:38:17.437287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics import roc_auc_score\n\n# Confirm both prediction matrices\nmri_p = mri_predictions\ntext_p = pred\n\nweights = np.arange(0.0, 1.01, 0.05)\n\nfusion_results = []\n\nfor w in weights:\n\n    fused = (\n        w * mri_p\n        + (1.0 - w) * text_p\n    )\n\n    aucs = []\n\n    for j, target in enumerate(label_cols):\n        auc = roc_auc_score(\n            Y[:, j],\n            fused[:, j]\n        )\n        aucs.append(auc)\n\n    macro = np.mean(aucs)\n\n    fusion_results.append(\n        (w, macro)\n    )\n\n# Sort by macro AUC\nfusion_results = sorted(\n    fusion_results,\n    key=lambda x: x[1],\n    reverse=True\n)\n\nprint(\"Top fusion weights:\")\nprint()\n\nfor w, score in fusion_results[:10]:\n    print(\n        f\"MRI weight: {w:.2f} | \"\n        f\"Report weight: {1-w:.2f} | \"\n        f\"Macro AUC: {score:.4f}\"\n    )\n\nbest_weight, best_auc = fusion_results[0]\n\nprint(\"\\nBest MRI weight:\", best_weight)\nprint(\"Best Report weight:\", 1 - best_weight)\nprint(\"Best Fusion Macro AUC:\", best_auc)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:39:00.984189Z","iopub.execute_input":"2026-08-09T08:39:00.984535Z","iopub.status.idle":"2026-08-09T08:39:01.411227Z","shell.execute_reply.started":"2026-08-09T08:39:00.984505Z","shell.execute_reply":"2026-08-09T08:39:01.409995Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"MRI tensor:\", mri_tensor.shape)\nprint(\"MRI embeddings:\", X_mri_embed.shape)\n\n# We need to know whether the per-series embeddings were retained.\nfor name, obj in list(globals().items()):\n    if isinstance(obj, np.ndarray):\n        print(name, obj.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:40:21.358833Z","iopub.execute_input":"2026-08-09T08:40:21.359247Z","iopub.status.idle":"2026-08-09T08:40:21.370795Z","shell.execute_reply.started":"2026-08-09T08:40:21.359212Z","shell.execute_reply":"2026-08-09T08:40:21.369451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\nprint(\"Known MRI embedding:\")\nprint(\"X_mri_embed:\", X_mri_embed.shape)\n\nprint(\"\\nMRI-related arrays currently in memory:\")\nfor name, obj in list(globals().items()):\n    if isinstance(obj, np.ndarray):\n        if \"mri\" in name.lower() or (\n            obj.ndim >= 3 and obj.shape[0] == 58\n        ):\n            print(f\"{name:30s} {obj.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:41:10.378915Z","iopub.execute_input":"2026-08-09T08:41:10.379287Z","iopub.status.idle":"2026-08-09T08:41:10.3906Z","shell.execute_reply.started":"2026-08-09T08:41:10.379249Z","shell.execute_reply":"2026-08-09T08:41:10.389373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport numpy as np\nfrom tqdm.auto import tqdm\n\n# Reuse the already-loaded model\nmodel.eval()\n\ndevice = torch.device(\"cpu\")\nmodel = model.to(device)\n\nn_studies, n_series, n_slices, H, W = X_mri.shape\n\nseries_embeddings = np.zeros(\n    (n_studies, n_series, 512),\n    dtype=np.float32\n)\n\nwith torch.no_grad():\n\n    for i in tqdm(range(n_studies), desc=\"Series embeddings\"):\n\n        # (5, 5, 224, 224)\n        study = X_mri[i]\n\n        # Each series: 5 grayscale slices\n        for s in range(n_series):\n\n            imgs = torch.tensor(\n                study[s],\n                dtype=torch.float32\n            )\n\n            # grayscale → 3 channels\n            imgs = imgs.unsqueeze(1).repeat(1, 3, 1, 1)\n\n            imgs = imgs.to(device)\n\n            # ResNet feature extractor\n            feats = model(imgs)\n\n            # Average the 5 slices\n            series_embeddings[i, s] = (\n                feats.mean(dim=0).cpu().numpy()\n            )\n\nprint(\"Series embedding shape:\", series_embeddings.shape)\nprint(\"Min:\", series_embeddings.min())\nprint(\"Max:\", series_embeddings.max())\nprint(\"Mean:\", series_embeddings.mean())\nprint(\"Std:\", series_embeddings.std())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:41:53.537319Z","iopub.execute_input":"2026-08-09T08:41:53.538449Z","iopub.status.idle":"2026-08-09T08:41:53.552808Z","shell.execute_reply.started":"2026-08-09T08:41:53.538406Z","shell.execute_reply":"2026-08-09T08:41:53.551257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torchvision.models as models\nimport numpy as np\nfrom tqdm.auto import tqdm\n\ndevice = torch.device(\"cpu\")\n\n# Create a fresh ResNet18\nresnet = models.resnet18(weights=models.ResNet18_Weights.DEFAULT)\n\n# Remove final classification layer\nresnet.fc = torch.nn.Identity()\n\nresnet = resnet.to(device)\nresnet.eval()\n\nprint(\"ResNet ready.\")\nprint(\"Output dimension:\", 512)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:42:48.170293Z","iopub.execute_input":"2026-08-09T08:42:48.170733Z","iopub.status.idle":"2026-08-09T08:42:48.3984Z","shell.execute_reply.started":"2026-08-09T08:42:48.170694Z","shell.execute_reply":"2026-08-09T08:42:48.397323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"series_embeddings = np.zeros(\n    (X_mri.shape[0], X_mri.shape[1], 512),\n    dtype=np.float32\n)\n\nwith torch.no_grad():\n\n    for i in tqdm(\n        range(X_mri.shape[0]),\n        desc=\"Series embeddings\"\n    ):\n\n        for s in range(X_mri.shape[1]):\n\n            # 5 slices × 224 × 224\n            imgs = torch.tensor(\n                X_mri[i, s],\n                dtype=torch.float32\n            )\n\n            # 5 × 1 × 224 × 224\n            imgs = imgs.unsqueeze(1)\n\n            # 5 × 3 × 224 × 224\n            imgs = imgs.repeat(1, 3, 1, 1)\n\n            imgs = imgs.to(device)\n\n            # 5 × 512\n            feats = resnet(imgs)\n\n            # Average slices within this series\n            series_embeddings[i, s] = (\n                feats.mean(dim=0)\n                .cpu()\n                .numpy()\n            )\n\nprint(\"Series embeddings:\", series_embeddings.shape)\nprint(\"Min:\", series_embeddings.min())\nprint(\"Max:\", series_embeddings.max())\nprint(\"Mean:\", series_embeddings.mean())\nprint(\"Std:\", series_embeddings.std())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:43:23.352954Z","iopub.execute_input":"2026-08-09T08:43:23.353383Z","iopub.status.idle":"2026-08-09T08:44:18.317946Z","shell.execute_reply.started":"2026-08-09T08:43:23.353348Z","shell.execute_reply":"2026-08-09T08:44:18.317097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nDATA = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\n\nprint(os.listdir(DATA))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-08T15:08:02.727237Z","iopub.execute_input":"2026-08-08T15:08:02.728148Z","iopub.status.idle":"2026-08-08T15:08:02.733806Z","shell.execute_reply.started":"2026-08-08T15:08:02.728114Z","shell.execute_reply":"2026-08-08T15:08:02.732697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Series embeddings:\", series_embeddings.shape)\n\n# Reconstruct the study-level embedding by averaging the 5 series\nstudy_embeddings_new = series_embeddings.mean(axis=1)\n\nprint(\"New study embeddings:\", study_embeddings_new.shape)\n\n# Compare against the original embedding\ndiff = np.abs(study_embeddings_new - X_mri_embed)\n\nprint(\"Mean absolute difference:\", diff.mean())\nprint(\"Maximum absolute difference:\", diff.max())\n\n# Correlation across all embedding values\ncorr = np.corrcoef(\n    study_embeddings_new.ravel(),\n    X_mri_embed.ravel()\n)[0, 1]\n\nprint(\"Correlation with original:\", corr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:45:44.600619Z","iopub.execute_input":"2026-08-09T08:45:44.601021Z","iopub.status.idle":"2026-08-09T08:45:44.618916Z","shell.execute_reply.started":"2026-08-09T08:45:44.600988Z","shell.execute_reply":"2026-08-09T08:45:44.617726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import LeaveOneOut\nfrom sklearn.metrics import roc_auc_score\n\n# Labels\nY = Y_mri.astype(float)\n\n# --------------------------------------------------\n# Representation A: average the 5 series\n# --------------------------------------------------\nX_series_mean = series_embeddings.mean(axis=1)\n\n# --------------------------------------------------\n# Representation B: preserve series identity\n# --------------------------------------------------\nX_series_flat = series_embeddings.reshape(\n    series_embeddings.shape[0], -1\n)\n\nprint(\"Series mean:\", X_series_mean.shape)\nprint(\"Series flat:\", X_series_flat.shape)\n\n\ndef loo_auc(X, Y, C=0.1):\n\n    loo = LeaveOneOut()\n    pred = np.zeros_like(Y, dtype=float)\n\n    for train_idx, val_idx in loo.split(X):\n\n        for j in range(Y.shape[1]):\n\n            y_train = Y[train_idx, j]\n\n            # Need both classes in training fold\n            if len(np.unique(y_train)) < 2:\n                pred[val_idx, j] = y_train.mean()\n                continue\n\n            clf = LogisticRegression(\n                C=C,\n                max_iter=2000,\n                class_weight=\"balanced\",\n                solver=\"liblinear\"\n            )\n\n            clf.fit(X[train_idx], y_train)\n            pred[val_idx, j] = clf.predict_proba(\n                X[val_idx]\n            )[:, 1]\n\n    aucs = {}\n\n    for j, col in enumerate(label_cols):\n        try:\n            aucs[col] = roc_auc_score(\n                Y[:, j],\n                pred[:, j]\n            )\n        except:\n            aucs[col] = np.nan\n\n    return aucs, pred\n\n\n# Test several regularization strengths\nfor C in [0.01, 0.03, 0.1, 0.3, 1.0]:\n\n    auc_mean, _ = loo_auc(\n        X_series_mean,\n        Y,\n        C=C\n    )\n\n    macro_mean = np.nanmean(list(auc_mean.values()))\n\n    auc_flat, _ = loo_auc(\n        X_series_flat,\n        Y,\n        C=C\n    )\n\n    macro_flat = np.nanmean(list(auc_flat.values()))\n\n    print(\n        f\"C={C:<4} | \"\n        f\"Series mean: {macro_mean:.4f} | \"\n        f\"Series flat: {macro_flat:.4f}\"\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:46:28.573308Z","iopub.execute_input":"2026-08-09T08:46:28.573816Z","iopub.status.idle":"2026-08-09T08:47:58.891737Z","shell.execute_reply.started":"2026-08-09T08:46:28.573773Z","shell.execute_reply":"2026-08-09T08:47:58.89057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = []\n\nfor C in [0.01, 0.03, 0.1, 0.3, 1.0]:\n\n    auc_mean, pred_mean = loo_auc(\n        X_series_mean,\n        Y,\n        C=C\n    )\n    macro_mean = np.nanmean(list(auc_mean.values()))\n\n    auc_flat, pred_flat = loo_auc(\n        X_series_flat,\n        Y,\n        C=C\n    )\n    macro_flat = np.nanmean(list(auc_flat.values()))\n\n    results.append({\n        \"C\": C,\n        \"Series_Mean_AUC\": macro_mean,\n        \"Series_Flat_AUC\": macro_flat\n    })\n\nresults_df = pd.DataFrame(results)\n\nprint(results_df.to_string(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:49:28.955262Z","iopub.execute_input":"2026-08-09T08:49:28.955629Z","iopub.status.idle":"2026-08-09T08:50:57.795234Z","shell.execute_reply.started":"2026-08-09T08:49:28.955574Z","shell.execute_reply":"2026-08-09T08:50:57.794109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compare the two 512-D representations directly\n\nprint(\"Original X_mri_embed:\")\nprint(\" shape:\", X_mri_embed.shape)\nprint(\" mean:\", X_mri_embed.mean())\nprint(\" std :\", X_mri_embed.std())\nprint(\" min :\", X_mri_embed.min())\nprint(\" max :\", X_mri_embed.max())\n\nprint(\"\\nNew series-mean:\")\nprint(\" shape:\", X_series_mean.shape)\nprint(\" mean:\", X_series_mean.mean())\nprint(\" std :\", X_series_mean.std())\nprint(\" min :\", X_series_mean.min())\nprint(\" max :\", X_series_mean.max())\n\n# Per-study cosine similarity\nfrom sklearn.metrics.pairwise import cosine_similarity\n\ncos_sim = np.diag(\n    cosine_similarity(\n        X_mri_embed,\n        X_series_mean\n    )\n)\n\nprint(\"\\nCosine similarity:\")\nprint(\"Mean:\", cos_sim.mean())\nprint(\"Median:\", np.median(cos_sim))\nprint(\"Min:\", cos_sim.min())\nprint(\"Max:\", cos_sim.max())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T09:21:00.492023Z","iopub.execute_input":"2026-08-09T09:21:00.492441Z","iopub.status.idle":"2026-08-09T09:21:00.511704Z","shell.execute_reply.started":"2026-08-09T09:21:00.492388Z","shell.execute_reply":"2026-08-09T09:21:00.510862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Find variables/functions that may contain the original MRI embedding pipeline\n\nfor name, obj in list(globals().items()):\n    name_l = name.lower()\n\n    if any(k in name_l for k in [\n        \"embed\",\n        \"resnet\",\n        \"extract\",\n        \"feature\",\n        \"mri\"\n    ]):\n        print(name, type(obj))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T09:21:45.681663Z","iopub.execute_input":"2026-08-09T09:21:45.682106Z","iopub.status.idle":"2026-08-09T09:21:45.690758Z","shell.execute_reply.started":"2026-08-09T09:21:45.682072Z","shell.execute_reply":"2026-08-09T09:21:45.689411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"X_mri_embed:\", X_mri_embed.shape)\n\n# Show currently defined callable functions\nfor name, obj in list(globals().items()):\n    if callable(obj) and not name.startswith(\"_\"):\n        print(\"FUNCTION:\", name)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T09:22:35.055756Z","iopub.execute_input":"2026-08-09T09:22:35.056157Z","iopub.status.idle":"2026-08-09T09:22:35.064139Z","shell.execute_reply.started":"2026-08-09T09:22:35.056129Z","shell.execute_reply":"2026-08-09T09:22:35.06293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import inspect\n\nprint(inspect.getsource(extract_study_embedding))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T09:23:22.93265Z","iopub.execute_input":"2026-08-09T09:23:22.933776Z","iopub.status.idle":"2026-08-09T09:23:22.941952Z","shell.execute_reply.started":"2026-08-09T09:23:22.93372Z","shell.execute_reply":"2026-08-09T09:23:22.940641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(inspect.getsource(extract_mri_features))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T09:24:09.897799Z","iopub.execute_input":"2026-08-09T09:24:09.898192Z","iopub.status.idle":"2026-08-09T09:24:09.908196Z","shell.execute_reply.started":"2026-08-09T09:24:09.898162Z","shell.execute_reply":"2026-08-09T09:24:09.906845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import inspect\n\nprint(inspect.getsource(extract_study_embedding))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T09:24:50.989254Z","iopub.execute_input":"2026-08-09T09:24:50.99023Z","iopub.status.idle":"2026-08-09T09:24:50.999377Z","shell.execute_reply.started":"2026-08-09T09:24:50.990186Z","shell.execute_reply":"2026-08-09T09:24:50.998121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Recreate the original study embeddings using the ORIGINAL function\n\nX_mri_embed_reproduced = []\n\nfor study_id in tqdm(all_mri_data.keys(), desc=\"Reproducing embeddings\"):\n\n    study_data = all_mri_data[study_id]\n\n    # all_mri_data may contain the series list directly\n    embedding = extract_study_embedding(study_data)\n\n    X_mri_embed_reproduced.append(embedding)\n\nX_mri_embed_reproduced = np.stack(X_mri_embed_reproduced)\n\nprint(\"Reproduced:\", X_mri_embed_reproduced.shape)\nprint(\"Original:\", X_mri_embed.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T09:27:23.112438Z","iopub.execute_input":"2026-08-09T09:27:23.11283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"diff = np.abs(\n    X_mri_embed_reproduced - X_mri_embed\n)\n\nprint(\"Mean absolute difference:\", diff.mean())\nprint(\"Maximum absolute difference:\", diff.max())\n\ncorr = np.corrcoef(\n    X_mri_embed_reproduced.ravel(),\n    X_mri_embed.ravel()\n)[0, 1]\n\nprint(\"Correlation:\", corr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T09:33:10.144567Z","iopub.execute_input":"2026-08-09T09:33:10.144962Z","iopub.status.idle":"2026-08-09T09:33:10.158205Z","shell.execute_reply.started":"2026-08-09T09:33:10.144927Z","shell.execute_reply":"2026-08-09T09:33:10.156614Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Keep the original winning baseline safe\n\nX_baseline = X_mri_embed.copy()\n\nprint(\"Baseline embedding:\", X_baseline.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T09:34:10.613131Z","iopub.execute_input":"2026-08-09T09:34:10.613499Z","iopub.status.idle":"2026-08-09T09:34:10.621104Z","shell.execute_reply.started":"2026-08-09T09:34:10.613468Z","shell.execute_reply":"2026-08-09T09:34:10.619912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"all_mri_data type:\", type(all_mri_data))\nprint(\"Number of studies:\", len(all_mri_data))\n\nfirst_id = list(all_mri_data.keys())[0]\n\nprint(\"\\nFirst study:\", first_id)\nprint(\"Type:\", type(all_mri_data[first_id]))\n\nif isinstance(all_mri_data[first_id], (list, tuple)):\n    print(\"Number of series:\", len(all_mri_data[first_id]))\n    print(\"Series shapes:\")\n    for i, s in enumerate(all_mri_data[first_id]):\n        print(i, np.asarray(s).shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T09:34:44.804858Z","iopub.execute_input":"2026-08-09T09:34:44.805204Z","iopub.status.idle":"2026-08-09T09:34:44.8131Z","shell.execute_reply.started":"2026-08-09T09:34:44.805177Z","shell.execute_reply":"2026-08-09T09:34:44.811903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Verify series embedding structure\n\nprint(\"Series embeddings shape:\", series_embeddings.shape)\n\nfor i, study_id in enumerate(all_mri_data.keys()):\n    n_series = len(all_mri_data[study_id])\n\n    if series_embeddings[i].shape[0] != n_series:\n        print(\n            \"MISMATCH:\",\n            i,\n            study_id,\n            \"MRI series =\", n_series,\n            \"embedding series =\", series_embeddings[i].shape[0]\n        )\n\nprint(\"Verification complete.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T09:36:40.704025Z","iopub.execute_input":"2026-08-09T09:36:40.704422Z","iopub.status.idle":"2026-08-09T09:36:40.713771Z","shell.execute_reply.started":"2026-08-09T09:36:40.70439Z","shell.execute_reply":"2026-08-09T09:36:40.712425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Show distribution of retained series embeddings\n\nseries_counts = [\n    len(all_mri_data[study_id])\n    for study_id in all_mri_data.keys()\n]\n\nprint(\"Series counts:\", series_counts)\nprint(\"Min:\", min(series_counts))\nprint(\"Max:\", max(series_counts))\nprint(\"Mean:\", np.mean(series_counts))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T09:37:41.04825Z","iopub.execute_input":"2026-08-09T09:37:41.048744Z","iopub.status.idle":"2026-08-09T09:37:41.056866Z","shell.execute_reply.started":"2026-08-09T09:37:41.048708Z","shell.execute_reply":"2026-08-09T09:37:41.055525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_all_series_embeddings(series_list):\n    \"\"\"\n    Preserve every available series embedding.\n\n    Input:\n        list of series\n        each series = (5, 224, 224)\n\n    Output:\n        (number_of_series, 512)\n    \"\"\"\n\n    all_embeddings = []\n\n    with torch.no_grad():\n\n        for series in series_list:\n\n            images = torch.from_numpy(\n                np.asarray(series)\n            ).float()\n\n            # (5, 224, 224) -> (5, 3, 224, 224)\n            images = images.unsqueeze(1).repeat(1, 3, 1, 1)\n\n            # Same ImageNet normalization as the winning baseline\n            images = (images - mean) / std\n\n            embeddings = resnet(images)\n\n            # Average 5 slices → one 512-D series embedding\n            series_embedding = embeddings.mean(dim=0)\n\n            all_embeddings.append(\n                series_embedding.cpu().numpy()\n            )\n\n    return np.stack(all_embeddings).astype(np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T09:40:55.165744Z","iopub.execute_input":"2026-08-09T09:40:55.166113Z","iopub.status.idle":"2026-08-09T09:40:55.173799Z","shell.execute_reply.started":"2026-08-09T09:40:55.166083Z","shell.execute_reply":"2026-08-09T09:40:55.172439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_series_embeddings = {}\n\nfor study_id, series_list in tqdm(\n    all_mri_data.items(),\n    desc=\"Extracting ALL series embeddings\"\n):\n    all_series_embeddings[study_id] = extract_all_series_embeddings(\n        series_list\n    )\n\nprint(\"Studies:\", len(all_series_embeddings))\n\nprint(\n    \"Series counts:\",\n    [v.shape[0] for v in all_series_embeddings.values()]\n)\n\nprint(\n    \"Embedding dimensions:\",\n    set(v.shape[1] for v in all_series_embeddings.values())\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T09:41:29.346056Z","iopub.execute_input":"2026-08-09T09:41:29.346892Z","iopub.status.idle":"2026-08-09T09:42:43.063909Z","shell.execute_reply.started":"2026-08-09T09:41:29.346852Z","shell.execute_reply":"2026-08-09T09:42:43.062562Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_all_series_mean = np.stack([\n    all_series_embeddings[study_id].mean(axis=0)\n    for study_id in all_mri_data.keys()\n])\n\nprint(\"Shape:\", X_all_series_mean.shape)\n\ndiff = np.abs(X_all_series_mean - X_mri_embed)\n\nprint(\"Mean absolute difference:\", diff.mean())\nprint(\"Maximum absolute difference:\", diff.max())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T09:49:02.053923Z","iopub.execute_input":"2026-08-09T09:49:02.05449Z","iopub.status.idle":"2026-08-09T09:49:02.07028Z","shell.execute_reply.started":"2026-08-09T09:49:02.054442Z","shell.execute_reply":"2026-08-09T09:49:02.068744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.model_selection import LeaveOneOut\nfrom sklearn.metrics import roc_auc_score\n\nstudy_ids = list(all_mri_data.keys())\n\n# Build pooled representations\nX_mean = np.stack([\n    all_series_embeddings[sid].mean(axis=0)\n    for sid in study_ids\n])\n\nX_max = np.stack([\n    all_series_embeddings[sid].max(axis=0)\n    for sid in study_ids\n])\n\nX_meanmax = np.concatenate(\n    [X_mean, X_max],\n    axis=1\n)\n\nprint(\"Mean:\", X_mean.shape)\nprint(\"Max :\", X_max.shape)\nprint(\"Mean+Max:\", X_meanmax.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T10:03:41.655307Z","iopub.execute_input":"2026-08-09T10:03:41.655729Z","iopub.status.idle":"2026-08-09T10:03:41.671435Z","shell.execute_reply.started":"2026-08-09T10:03:41.655688Z","shell.execute_reply":"2026-08-09T10:03:41.670018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nDATA = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\n\ntrain = pd.read_csv(f\"{DATA}/train.csv\")\n\nprint(\"Number of training studies:\", len(train))\nprint(\"Columns:\")\nprint(train.columns.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-08T15:09:23.397299Z","iopub.execute_input":"2026-08-08T15:09:23.397629Z","iopub.status.idle":"2026-08-08T15:09:24.832351Z","shell.execute_reply.started":"2026-08-08T15:09:23.39758Z","shell.execute_reply":"2026-08-08T15:09:24.831151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_cols = [\n    \"ACL\",\n    \"MCL\",\n    \"Medial Meniscus\",\n    \"Lateral Meniscus\",\n    \"Medial OA\",\n    \"Lateral OA\",\n    \"PF OA\",\n    \"Effusion\",\n    \"Synovitis\",\n    \"Baker's\",\n    \"Contusion\",\n    \"Fracture\"\n]\n\nprint(\"Positive cases per abnormality:\")\nprint(train[label_cols].sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-08T15:10:39.287564Z","iopub.execute_input":"2026-08-08T15:10:39.288341Z","iopub.status.idle":"2026-08-08T15:10:39.298261Z","shell.execute_reply.started":"2026-08-08T15:10:39.288305Z","shell.execute_reply":"2026-08-08T15:10:39.297272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train[\"Report\"].iloc[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:25:24.020731Z","iopub.execute_input":"2026-08-09T07:25:24.021234Z","iopub.status.idle":"2026-08-09T07:25:24.036869Z","shell.execute_reply.started":"2026-08-09T07:25:24.0212Z","shell.execute_reply":"2026-08-09T07:25:24.035552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nDATA = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\n\ntrain = pd.read_csv(f\"{DATA}/train.csv\")\n\nprint(\"Training studies:\", len(train))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:25:04.576184Z","iopub.execute_input":"2026-08-09T07:25:04.576559Z","iopub.status.idle":"2026-08-09T07:25:06.507202Z","shell.execute_reply.started":"2026-08-09T07:25:04.576525Z","shell.execute_reply":"2026-08-09T07:25:06.506042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in range(5):\n    print(\"=\" * 80)\n    print(f\"REPORT {i}\")\n    print(train[\"Report\"].iloc[i])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:31:50.193404Z","iopub.execute_input":"2026-08-09T07:31:50.194318Z","iopub.status.idle":"2026-08-09T07:31:50.202159Z","shell.execute_reply.started":"2026-08-09T07:31:50.19428Z","shell.execute_reply":"2026-08-09T07:31:50.201266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_series = pd.read_csv(f\"{DATA}/train_series.csv\")\n\nprint(\"Training series:\", len(train_series))\nprint(train_series.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:33:30.92828Z","iopub.execute_input":"2026-08-09T07:33:30.928741Z","iopub.status.idle":"2026-08-09T07:33:31.063204Z","shell.execute_reply.started":"2026-08-09T07:33:30.928706Z","shell.execute_reply":"2026-08-09T07:33:31.062135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import glob\n\ndicom_files = glob.glob(\n    f\"{DATA}/train_series/**/*.dcm\",\n    recursive=True\n)\n\nprint(\"DICOM files found:\", len(dicom_files))\nprint(\"First DICOM:\")\nprint(dicom_files[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:37:32.58019Z","iopub.execute_input":"2026-08-09T07:37:32.580504Z","iopub.status.idle":"2026-08-09T07:38:12.301287Z","shell.execute_reply.started":"2026-08-09T07:37:32.580477Z","shell.execute_reply":"2026-08-09T07:38:12.299608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"study_id = train_series[\"StudyInstanceUID\"].iloc[0]\nseries_id = train_series[\"SeriesInstanceUID\"].iloc[0]\n\nprint(\"Study:\", study_id)\nprint(\"Series:\", series_id)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:40:15.469449Z","iopub.execute_input":"2026-08-09T07:40:15.469825Z","iopub.status.idle":"2026-08-09T07:40:15.477532Z","shell.execute_reply.started":"2026-08-09T07:40:15.469797Z","shell.execute_reply":"2026-08-09T07:40:15.47623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nseries_path = os.path.join(\n    DATA,\n    \"train_series\",\n    study_id,\n    series_id\n)\n\nprint(series_path)\nprint(\"Number of files:\", len(os.listdir(series_path)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:41:08.651065Z","iopub.execute_input":"2026-08-09T07:41:08.651387Z","iopub.status.idle":"2026-08-09T07:41:08.665278Z","shell.execute_reply.started":"2026-08-09T07:41:08.651362Z","shell.execute_reply":"2026-08-09T07:41:08.664008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nimport os\n\ndicom_file = os.path.join(series_path, os.listdir(series_path)[0])\n\nds = pydicom.dcmread(dicom_file)\n\nprint(\"DICOM file:\", dicom_file)\nprint(\"Image shape:\", ds.pixel_array.shape)\nprint(\"Data type:\", ds.pixel_array.dtype)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:42:04.603572Z","iopub.execute_input":"2026-08-09T07:42:04.604021Z","iopub.status.idle":"2026-08-09T07:42:05.941909Z","shell.execute_reply.started":"2026-08-09T07:42:04.603989Z","shell.execute_reply":"2026-08-09T07:42:05.940657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(6, 6))\nplt.imshow(ds.pixel_array, cmap=\"gray\")\nplt.axis(\"off\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:42:44.430213Z","iopub.execute_input":"2026-08-09T07:42:44.430536Z","iopub.status.idle":"2026-08-09T07:42:44.761398Z","shell.execute_reply.started":"2026-08-09T07:42:44.430511Z","shell.execute_reply":"2026-08-09T07:42:44.76021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"study_series = train_series[\n    train_series[\"StudyInstanceUID\"] == study_id\n]\n\nprint(\"Number of series:\", len(study_series))\ndisplay(study_series)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:45:16.161143Z","iopub.execute_input":"2026-08-09T07:45:16.161549Z","iopub.status.idle":"2026-08-09T07:45:16.206637Z","shell.execute_reply.started":"2026-08-09T07:45:16.161517Z","shell.execute_reply":"2026-08-09T07:45:16.205265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nfor _, row in study_series.iterrows():\n    path = os.path.join(\n        DATA,\n        \"train_series\",\n        row[\"StudyInstanceUID\"],\n        row[\"SeriesInstanceUID\"]\n    )\n\n    n_slices = len(os.listdir(path))\n\n    print(\n        row[\"Anatomical_Plane\"],\n        \"| Fluid:\", row[\"Fluid_Sensitive\"],\n        \"| Fat suppression:\", row[\"Fat_Suppression\"],\n        \"| Slices:\", n_slices\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:46:10.278712Z","iopub.execute_input":"2026-08-09T07:46:10.279227Z","iopub.status.idle":"2026-08-09T07:46:10.320979Z","shell.execute_reply.started":"2026-08-09T07:46:10.279189Z","shell.execute_reply":"2026-08-09T07:46:10.319064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pydicom\nimport matplotlib.pyplot as plt\n\nfor _, row in study_series.iterrows():\n\n    path = os.path.join(\n        DATA,\n        \"train_series\",\n        row[\"StudyInstanceUID\"],\n        row[\"SeriesInstanceUID\"]\n    )\n\n    files = sorted(os.listdir(path))\n    middle_file = files[len(files) // 2]\n\n    ds_temp = pydicom.dcmread(os.path.join(path, middle_file))\n\n    plt.figure(figsize=(5, 5))\n    plt.imshow(ds_temp.pixel_array, cmap=\"gray\")\n    plt.title(\n        f\"{row['Anatomical_Plane']} | \"\n        f\"Fluid={row['Fluid_Sensitive']} | \"\n        f\"FatSupp={row['Fat_Suppression']}\"\n    )\n    plt.axis(\"off\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:48:18.605829Z","iopub.execute_input":"2026-08-09T07:48:18.606318Z","iopub.status.idle":"2026-08-09T07:48:19.871114Z","shell.execute_reply.started":"2026-08-09T07:48:18.606284Z","shell.execute_reply":"2026-08-09T07:48:19.869337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pydicom\n\nfor _, row in study_series.iterrows():\n\n    path = os.path.join(\n        DATA,\n        \"train_series\",\n        row[\"StudyInstanceUID\"],\n        row[\"SeriesInstanceUID\"]\n    )\n\n    files = os.listdir(path)\n\n    ds_first = pydicom.dcmread(os.path.join(path, files[0]), stop_before_pixels=True)\n\n    print(\"=\" * 60)\n    print(\"Plane:\", row[\"Anatomical_Plane\"])\n    print(\"Slices:\", len(files))\n    print(\"Image Position:\", getattr(ds_first, \"ImagePositionPatient\", \"N/A\"))\n    print(\"Slice Location:\", getattr(ds_first, \"SliceLocation\", \"N/A\"))\n    print(\"Instance Number:\", getattr(ds_first, \"InstanceNumber\", \"N/A\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:49:30.134973Z","iopub.execute_input":"2026-08-09T07:49:30.135323Z","iopub.status.idle":"2026-08-09T07:49:30.192813Z","shell.execute_reply.started":"2026-08-09T07:49:30.135292Z","shell.execute_reply":"2026-08-09T07:49:30.191607Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pydicom\nimport numpy as np\n\ndef load_series(series_path):\n    slices = []\n\n    for filename in os.listdir(series_path):\n        filepath = os.path.join(series_path, filename)\n\n        ds = pydicom.dcmread(filepath)\n\n        if hasattr(ds, \"ImagePositionPatient\"):\n            position = np.array(ds.ImagePositionPatient, dtype=float)\n        else:\n            position = np.array([0, 0, getattr(ds, \"InstanceNumber\", 0)])\n\n        slices.append((position, ds))\n\n    # Determine the direction in which slices vary most\n    positions = np.array([x[0] for x in slices])\n\n    axis = np.argmax(np.ptp(positions, axis=0))\n\n    # Sort along that anatomical axis\n    slices.sort(key=lambda x: x[0][axis])\n\n    return [x[1] for x in slices]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:51:55.445243Z","iopub.execute_input":"2026-08-09T07:51:55.445676Z","iopub.status.idle":"2026-08-09T07:51:55.456284Z","shell.execute_reply.started":"2026-08-09T07:51:55.445645Z","shell.execute_reply":"2026-08-09T07:51:55.454664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_series_path = os.path.join(\n    DATA,\n    \"train_series\",\n    study_id,\n    series_id\n)\n\nordered_slices = load_series(test_series_path)\n\nprint(\"Ordered slices:\", len(ordered_slices))\nprint(\"First position:\", ordered_slices[0].ImagePositionPatient)\nprint(\"Last position:\", ordered_slices[-1].ImagePositionPatient)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:53:18.841294Z","iopub.execute_input":"2026-08-09T07:53:18.841871Z","iopub.status.idle":"2026-08-09T07:53:19.066459Z","shell.execute_reply.started":"2026-08-09T07:53:18.84183Z","shell.execute_reply":"2026-08-09T07:53:19.065207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ds_check = ordered_slices[len(ordered_slices) // 2]\n\nprint(\"ImageOrientationPatient:\")\nprint(ds_check.ImageOrientationPatient)\n\nprint(\"\\nPixelSpacing:\")\nprint(ds_check.PixelSpacing)\n\nprint(\"\\nRows:\", ds_check.Rows)\nprint(\"Columns:\", ds_check.Columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:54:14.736389Z","iopub.execute_input":"2026-08-09T07:54:14.738537Z","iopub.status.idle":"2026-08-09T07:54:14.74677Z","shell.execute_reply.started":"2026-08-09T07:54:14.73849Z","shell.execute_reply":"2026-08-09T07:54:14.745543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img = ordered_slices[len(ordered_slices) // 2].pixel_array\n\nprint(\"Minimum:\", img.min())\nprint(\"Maximum:\", img.max())\nprint(\"Mean:\", img.mean())\nprint(\"Standard deviation:\", img.std())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:55:04.932372Z","iopub.execute_input":"2026-08-09T07:55:04.932824Z","iopub.status.idle":"2026-08-09T07:55:04.950326Z","shell.execute_reply.started":"2026-08-09T07:55:04.932788Z","shell.execute_reply":"2026-08-09T07:55:04.947767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\ndef normalize_mri(ds):\n    img = ds.pixel_array.astype(np.float32)\n\n    # Robust percentile window\n    low, high = np.percentile(img, [1, 99])\n\n    img = np.clip(img, low, high)\n\n    # Scale to 0–1\n    img = (img - low) / (high - low + 1e-8)\n\n    return img\n\nnormalized = normalize_mri(\n    ordered_slices[len(ordered_slices) // 2]\n)\n\nprint(\"Minimum:\", normalized.min())\nprint(\"Maximum:\", normalized.max())\nprint(\"Mean:\", normalized.mean())\nprint(\"Standard deviation:\", normalized.std())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:56:55.597155Z","iopub.execute_input":"2026-08-09T07:56:55.597526Z","iopub.status.idle":"2026-08-09T07:56:55.629222Z","shell.execute_reply.started":"2026-08-09T07:56:55.597496Z","shell.execute_reply":"2026-08-09T07:56:55.628003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\nimport numpy as np\nimport matplotlib.pyplot as plt\n\ndef resize_mri(img, size=(224, 224)):\n    img_uint8 = (img * 255).clip(0, 255).astype(np.uint8)\n    resized = Image.fromarray(img_uint8).resize(\n        size,\n        Image.Resampling.BILINEAR\n    )\n    return np.asarray(resized).astype(np.float32) / 255.0\n\nresized = resize_mri(normalized)\n\nprint(\"Original:\", normalized.shape)\nprint(\"Resized:\", resized.shape)\n\nplt.figure(figsize=(6, 6))\nplt.imshow(resized, cmap=\"gray\")\nplt.axis(\"off\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:57:48.446401Z","iopub.execute_input":"2026-08-09T07:57:48.446757Z","iopub.status.idle":"2026-08-09T07:57:48.620273Z","shell.execute_reply.started":"2026-08-09T07:57:48.446728Z","shell.execute_reply":"2026-08-09T07:57:48.619389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Original:\", normalized.shape)\nprint(\"Resized:\", resized.shape)\n\nplt.figure(figsize=(6, 6))\nplt.imshow(resized, cmap=\"gray\")\nplt.axis(\"off\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:59:21.898891Z","iopub.execute_input":"2026-08-09T07:59:21.899245Z","iopub.status.idle":"2026-08-09T07:59:22.062663Z","shell.execute_reply.started":"2026-08-09T07:59:21.899218Z","shell.execute_reply":"2026-08-09T07:59:22.06155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pydicom\nimport numpy as np\n\ndef get_series_images(study_id, series_id, n_slices=5):\n    series_path = os.path.join(\n        DATA,\n        \"train_series\",\n        study_id,\n        series_id\n    )\n\n    # Load and spatially order the series\n    ds_list = load_series(series_path)\n\n    # Pick evenly spaced slices\n    indices = np.linspace(\n        0,\n        len(ds_list) - 1,\n        n_slices\n    ).astype(int)\n\n    images = []\n\n    for idx in indices:\n        img = normalize_mri(ds_list[idx])\n        img = resize_mri(img, (224, 224))\n        images.append(img)\n\n    return np.stack(images)\n\n\nstudy_images = []\n\nfor _, row in study_series.iterrows():\n    imgs = get_series_images(\n        row[\"StudyInstanceUID\"],\n        row[\"SeriesInstanceUID\"],\n        n_slices=5\n    )\n    study_images.append(imgs)\n\nstudy_images = np.stack(study_images)\n\nprint(\"Study tensor shape:\", study_images.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:00:00.007301Z","iopub.execute_input":"2026-08-09T08:00:00.007647Z","iopub.status.idle":"2026-08-09T08:00:01.064661Z","shell.execute_reply.started":"2026-08-09T08:00:00.007621Z","shell.execute_reply":"2026-08-09T08:00:01.063366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Take the first 20 training studies\nprototype_studies = train[\"StudyInstanceUID\"].iloc[:20].tolist()\n\nprint(\"Prototype studies:\", len(prototype_studies))\nprint(prototype_studies[:3])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:00:58.187532Z","iopub.execute_input":"2026-08-09T08:00:58.187905Z","iopub.status.idle":"2026-08-09T08:00:58.194972Z","shell.execute_reply.started":"2026-08-09T08:00:58.187877Z","shell.execute_reply":"2026-08-09T08:00:58.193603Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_cols = [\n    \"ACL\",\n    \"MCL\",\n    \"Medial Meniscus\",\n    \"Lateral Meniscus\",\n    \"Medial OA\",\n    \"Lateral OA\",\n    \"PF OA\",\n    \"Effusion\",\n    \"Synovitis\",\n    \"Baker's\",\n    \"Contusion\",\n    \"Fracture\"\n]\n\nprototype_labels = train[\n    train[\"StudyInstanceUID\"].isin(prototype_studies)\n][[\"StudyInstanceUID\"] + target_cols].copy()\n\nprint(\"Prototype label shape:\", prototype_labels.shape)\n\ndisplay(prototype_labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:01:48.609434Z","iopub.execute_input":"2026-08-09T08:01:48.610526Z","iopub.status.idle":"2026-08-09T08:01:48.649907Z","shell.execute_reply.started":"2026-08-09T08:01:48.610488Z","shell.execute_reply":"2026-08-09T08:01:48.648696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ntrain_root = f\"{DATA}/train_series\"\n\nstudy = os.listdir(train_root)[0]\nstudy_path = os.path.join(train_root, study)\n\nseries = os.listdir(study_path)[0]\nseries_path = os.path.join(study_path, series)\n\ndicom = os.listdir(series_path)[0]\ndicom_path = os.path.join(series_path, dicom)\n\nprint(dicom_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T07:38:12.306275Z","iopub.status.idle":"2026-08-09T07:38:12.306658Z","shell.execute_reply.started":"2026-08-09T07:38:12.306442Z","shell.execute_reply":"2026-08-09T07:38:12.306465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Total studies:\", len(train))\n\nprint(\"\\nLabeled studies per abnormality:\")\nprint(train[target_cols].notna().sum())\n\nprint(\"\\nStudies with at least one label:\",\n      train[target_cols].notna().any(axis=1).sum())\n\nprint(\"\\nStudies with all 12 labels:\",\n      train[target_cols].notna().all(axis=1).sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:05:18.506855Z","iopub.execute_input":"2026-08-09T08:05:18.507285Z","iopub.status.idle":"2026-08-09T08:05:18.52733Z","shell.execute_reply.started":"2026-08-09T08:05:18.507255Z","shell.execute_reply":"2026-08-09T08:05:18.526335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labeled_train = train[\n    train[target_cols].notna().all(axis=1)\n].copy()\n\nprint(\"Labeled studies:\", len(labeled_train))\n\nprint(\"\\nPositive cases:\")\nprint(labeled_train[target_cols].sum())\n\nprint(\"\\nNegative cases:\")\nprint((1 - labeled_train[target_cols]).sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:05:58.219434Z","iopub.execute_input":"2026-08-09T08:05:58.220706Z","iopub.status.idle":"2026-08-09T08:05:58.237491Z","shell.execute_reply.started":"2026-08-09T08:05:58.220661Z","shell.execute_reply":"2026-08-09T08:05:58.236356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i, row in labeled_train.head(10).iterrows():\n    print(\"=\" * 100)\n    print(\"STUDY:\", row[\"StudyInstanceUID\"])\n    print(\"\\nREPORT:\")\n    print(row[\"Report\"][:2000])\n    print(\"\\nLABELS:\")\n    print(row[target_cols].to_dict())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T08:07:05.580004Z","iopub.execute_input":"2026-08-09T08:07:05.580405Z","iopub.status.idle":"2026-08-09T08:07:05.597807Z","shell.execute_reply.started":"2026-08-09T08:07:05.580365Z","shell.execute_reply":"2026-08-09T08:07:05.596666Z"}},"outputs":[],"execution_count":null}]}