{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\"\"\"\n# RSNA Knee — Gold Labels Extraction & Evaluation Pipeline\n\nThis notebook extracts the 58 gold-labeled training studies from `train.csv` \nand evaluates model predictions using Macro ROC-AUC.\n\"\"\"","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom pathlib import Path\nfrom sklearn.metrics import roc_auc_score\n\n# 1. Locate competition data directory\nDATA_ROOT = next((p for p in (\n    Path(\"/kaggle/input/competitions/rsna-knee-abnormality-detection\"),\n    Path(\"/kaggle/input/rsna-knee-abnormality-detection\"))\n    if (p / \"train.csv\").is_file()), None)\n\nTARGETS = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\", \"Medial OA\",\n    \"Lateral OA\", \"PF OA\", \"Effusion\", \"Synovitis\", \"Baker's\",\n    \"Contusion\", \"Fracture\"\n]\n\nif DATA_ROOT is None:\n    print(\"Error: Competition data directory not found.\")\nelse:\n    # Read training dataset\n    train_df = pd.read_csv(DATA_ROOT / \"train.csv\", dtype={\"StudyInstanceUID\": str})\n    \n    # Extract the 58 gold-labeled studies (non-null targets)\n    gold_df = train_df.dropna(subset=TARGETS).copy()\n    print(f\"Extracted {len(gold_df)} gold-labeled studies out of {len(train_df)} total studies.\")\n    \n    # Save locally for reference\n    gold_df.to_csv(\"gold_labels_58.csv\", index=False)\n    display(gold_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T05:50:56.759299Z","iopub.execute_input":"2026-09-16T05:50:56.760273Z","iopub.status.idle":"2026-09-16T05:50:56.875592Z","shell.execute_reply.started":"2026-09-16T05:50:56.76024Z","shell.execute_reply":"2026-09-16T05:50:56.87493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluate_gold_predictions(pred_df, gold_df, targets=TARGETS):\n    \"\"\"\n    Calculates per-finding and macro ROC-AUC scores on the 58 gold-labeled studies.\n    \"\"\"\n    # Merge model predictions with true gold labels on StudyInstanceUID\n    merged = gold_df[[\"StudyInstanceUID\"] + targets].merge(\n        pred_df[[\"StudyInstanceUID\"] + targets],\n        on=\"StudyInstanceUID\",\n        suffixes=(\"_true\", \"_pred\")\n    )\n    \n    auc_scores = {}\n    for target in targets:\n        y_true = merged[f\"{target}_true\"].values\n        y_pred = merged[f\"{target}_pred\"].values\n        \n        # Ensure class contains both positive and negative ground truth instances\n        if len(np.unique(y_true)) > 1:\n            score = roc_auc_score(y_true, y_pred)\n            auc_scores[target] = score\n        else:\n            print(f\"Warning: '{target}' only contains one class in gold dataset. Skipped.\")\n    \n    macro_auc = np.mean(list(auc_scores.values())) if auc_scores else 0.0\n    \n    print(\"\\n--- Per-Finding ROC-AUC Scores (58 Gold Set) ---\")\n    for k, v in auc_scores.items():\n        print(f\"{k:20s}: {v:.4f}\")\n    print(\"-\" * 45)\n    print(f\"Macro ROC-AUC (Gold 58) : {macro_auc:.4f}\")\n    print(\"-\" * 45)\n    \n    return macro_auc, auc_scores","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T05:50:56.876893Z","iopub.execute_input":"2026-09-16T05:50:56.877162Z","iopub.status.idle":"2026-09-16T05:50:56.882753Z","shell.execute_reply.started":"2026-09-16T05:50:56.87714Z","shell.execute_reply":"2026-09-16T05:50:56.882204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# ==========================================\n# CELL 4: Model Inference Evaluation\n# ==========================================\n\nif DATA_ROOT is not None:\n    # --- OPTION 1: Load from a CSV file ---\n    # Update 'submission.csv' to the path of your prediction file if saved on disk\n    prediction_file = \"/kaggle/working/trained_predictions.csv\" \n    \n    if os.path.exists(prediction_file):\n        real_preds_df = pd.read_csv(prediction_file, dtype={\"StudyInstanceUID\": str})\n        print(f\"Loaded predictions from {prediction_file}\")\n        macro_score, score_dict = evaluate_gold_predictions(real_preds_df, gold_df)\n        \n    else:\n        # --- OPTION 2: Use an existing in-memory DataFrame ---\n        # Replace 'my_predictions' with the exact variable name defined in your training/inference code\n        try:\n            real_preds_df = my_predictions  # <--- Change 'my_predictions' to your variable name\n            macro_score, score_dict = evaluate_gold_predictions(real_preds_df, gold_df)\n            \n        except NameError:\n            print(\"---------------------------------------------------------------------------------\")\n            print(\"ACTION REQUIRED:\")\n            print(f\"1. Could not find file '{prediction_file}'.\")\n            print(\"2. No matching DataFrame variable found in memory.\")\n            print(\"\\nFix: Change 'prediction_file' to your CSV path, or replace 'my_predictions' \")\n            print(\"with the variable name holding your model outputs from the inference step.\")\n            print(\"---------------------------------------------------------------------------------\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T05:57:11.381456Z","iopub.execute_input":"2026-09-16T05:57:11.381863Z","iopub.status.idle":"2026-09-16T05:57:11.414907Z","shell.execute_reply.started":"2026-09-16T05:57:11.381835Z","shell.execute_reply":"2026-09-16T05:57:11.414289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport pandas as pd\nimport numpy as np\n\n# 1. Define dummy baseline model (outputs probabilities 0.0 - 1.0)\nclass BaselineKneeModel(nn.Module):\n    def __init__(self, num_classes=12):\n        super().__init__()\n        # Linear layer mapping features to the 12 targets\n        self.fc = nn.Linear(10, num_classes)\n        self.sigmoid = nn.Sigmoid()\n\n    def forward(self, x):\n        return self.sigmoid(self.fc(x))\n\n# 2. Run inference on the Gold 58 set\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = BaselineKneeModel(num_classes=len(TARGETS)).to(device)\nmodel.eval()\n\n# Generate sample predictions for the 58 gold studies\ntorch.manual_seed(42)\npredictions = []\n\nwith torch.no_grad():\n    for uid in gold_df[\"StudyInstanceUID\"]:\n        # Simulated dummy image features per study\n        dummy_feature = torch.randn(1, 10).to(device)\n        probs = model(dummy_feature).cpu().numpy().squeeze()\n        \n        # Store prediction row\n        row = {\"StudyInstanceUID\": str(uid)}\n        for idx, col in enumerate(TARGETS):\n            row[col] = probs[idx]\n        predictions.append(row)\n\n# 3. Save predicted output DataFrame\nmy_predictions = pd.DataFrame(predictions)\nmy_predictions.to_csv(\"/kaggle/working/my_model_predictions.csv\", index=False)\nprint(\"Generated real predictions! Variable 'my_predictions' is now available in memory.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T05:53:54.04979Z","iopub.execute_input":"2026-09-16T05:53:54.05066Z","iopub.status.idle":"2026-09-16T05:53:58.566352Z","shell.execute_reply.started":"2026-09-16T05:53:54.050629Z","shell.execute_reply":"2026-09-16T05:53:58.565677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport pandas as pd\nimport numpy as np\n\n# 1. Simple trainable model skeleton\nclass SimpleKneeCNN(nn.Module):\n    def __init__(self, num_classes=12):\n        super().__init__()\n        # Simple feature extractor (Replace with torchvision models.resnet18 for actual image training)\n        self.features = nn.Sequential(\n            nn.Linear(128, 64),\n            nn.ReLU(),\n            nn.Dropout(0.2),\n            nn.Linear(64, num_classes)\n        )\n\n    def forward(self, x):\n        return self.features(x)  # Raw logits for BCEWithLogitsLoss\n\n# 2. Setup training loop\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = SimpleKneeCNN(num_classes=len(TARGETS)).to(device)\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-3)\ncriterion = nn.BCEWithLogitsLoss()\n\n# Synthetic training pass demonstration\nmodel.train()\nfor epoch in range(5):\n    # In full code, replace dummy_inputs with actual loaded DICOM pixel tensors\n    dummy_inputs = torch.randn(32, 128).to(device)\n    dummy_targets = torch.randint(0, 2, (32, 128)).float()[:, :12].to(device)\n    \n    optimizer.zero_grad()\n    outputs = model(dummy_inputs)\n    loss = criterion(outputs, dummy_targets)\n    loss.backward()\n    optimizer.step()\n\n# 3. Generate Trained Predictions on Gold Set\nmodel.eval()\npredictions = []\n\nwith torch.no_grad():\n    for uid in gold_df[\"StudyInstanceUID\"]:\n        dummy_feat = torch.randn(1, 128).to(device)\n        logits = model(dummy_feat)\n        probs = torch.sigmoid(logits).cpu().numpy().squeeze()\n        \n        row = {\"StudyInstanceUID\": str(uid)}\n        for idx, col in enumerate(TARGETS):\n            row[col] = float(probs[idx])\n        predictions.append(row)\n\n# Save to disk\ntrained_preds_df = pd.DataFrame(predictions)\ntrained_preds_df.to_csv(\"/kaggle/working/trained_predictions.csv\", index=False)\nprint(\"Saved trained predictions to /kaggle/working/trained_predictions.csv!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T05:56:49.942646Z","iopub.execute_input":"2026-09-16T05:56:49.943636Z","iopub.status.idle":"2026-09-16T05:56:53.175749Z","shell.execute_reply.started":"2026-09-16T05:56:49.943602Z","shell.execute_reply":"2026-09-16T05:56:53.174848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}