{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\n\ncompetition_path = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\n\nprint(\"Dataset found:\", os.path.exists(competition_path))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.44973Z","iopub.execute_input":"2026-08-10T14:44:11.450257Z","iopub.status.idle":"2026-08-10T14:44:11.461324Z","shell.execute_reply.started":"2026-08-10T14:44:11.450221Z","shell.execute_reply":"2026-08-10T14:44:11.460464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ninput_path = \"/kaggle/input\"\n\nfor item in os.listdir(input_path):\n    full_path = os.path.join(input_path, item)\n    print(item, \"📁\" if os.path.isdir(full_path) else \"📄\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.462638Z","iopub.execute_input":"2026-08-10T14:44:11.462986Z","iopub.status.idle":"2026-08-10T14:44:11.468761Z","shell.execute_reply.started":"2026-08-10T14:44:11.46296Z","shell.execute_reply":"2026-08-10T14:44:11.467952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\npath = \"/kaggle/input/competitions\"\n\nfor item in os.listdir(path):\n    print(item)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.469697Z","iopub.execute_input":"2026-08-10T14:44:11.469974Z","iopub.status.idle":"2026-08-10T14:44:11.484157Z","shell.execute_reply.started":"2026-08-10T14:44:11.469943Z","shell.execute_reply":"2026-08-10T14:44:11.483281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"competition_path = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\n\nfor item in os.listdir(competition_path):\n    full_path = os.path.join(competition_path, item)\n\n    if os.path.isdir(full_path):\n        print(\"📁\", item)\n    else:\n        print(\"📄\", item)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.485694Z","iopub.execute_input":"2026-08-10T14:44:11.485923Z","iopub.status.idle":"2026-08-10T14:44:11.503527Z","shell.execute_reply.started":"2026-08-10T14:44:11.485903Z","shell.execute_reply":"2026-08-10T14:44:11.502611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(os.path.join(competition_path, \"train.csv\"))\ntest = pd.read_csv(os.path.join(competition_path, \"test.csv\"))\ntrain_series = pd.read_csv(os.path.join(competition_path, \"train_series.csv\"))\ntest_series = pd.read_csv(os.path.join(competition_path, \"test_series.csv\"))\nsample_submission = pd.read_csv(\n    os.path.join(competition_path, \"sample_submission.csv\")\n)\n\nprint(\"train:\", train.shape)\nprint(\"test:\", test.shape)\nprint(\"train_series:\", train_series.shape)\nprint(\"test_series:\", test_series.shape)\nprint(\"sample_submission:\", sample_submission.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.504479Z","iopub.execute_input":"2026-08-10T14:44:11.504806Z","iopub.status.idle":"2026-08-10T14:44:11.661785Z","shell.execute_reply.started":"2026-08-10T14:44:11.504776Z","shell.execute_reply":"2026-08-10T14:44:11.661136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"TRAIN COLUMNS:\")\nprint(train.columns.tolist())\n\nprint(\"\\nTEST COLUMNS:\")\nprint(test.columns.tolist())\n\nprint(\"\\nTRAIN SERIES COLUMNS:\")\nprint(train_series.columns.tolist())\n\nprint(\"\\nSAMPLE SUBMISSION COLUMNS:\")\nprint(sample_submission.columns.tolist())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.662829Z","iopub.execute_input":"2026-08-10T14:44:11.663157Z","iopub.status.idle":"2026-08-10T14:44:11.66857Z","shell.execute_reply.started":"2026-08-10T14:44:11.66312Z","shell.execute_reply":"2026-08-10T14:44:11.667537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option(\"display.max_columns\", None)\n\ntrain.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.6695Z","iopub.execute_input":"2026-08-10T14:44:11.669822Z","iopub.status.idle":"2026-08-10T14:44:11.692042Z","shell.execute_reply.started":"2026-08-10T14:44:11.669773Z","shell.execute_reply":"2026-08-10T14:44:11.691142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.iloc[:, 2:].sum().sort_values(ascending=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.693297Z","iopub.execute_input":"2026-08-10T14:44:11.693671Z","iopub.status.idle":"2026-08-10T14:44:11.710011Z","shell.execute_reply.started":"2026-08-10T14:44:11.693634Z","shell.execute_reply":"2026-08-10T14:44:11.70887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_columns = [\n    'ACL', 'MCL', 'Medial Meniscus', 'Lateral Meniscus',\n    'Medial OA', 'Lateral OA', 'PF OA', 'Effusion',\n    'Synovitis', \"Baker's\", 'Contusion', 'Fracture'\n]\n\nfor column in target_columns:\n    print(f\"\\n{column}:\")\n    print(train[column].value_counts(dropna=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.711402Z","iopub.execute_input":"2026-08-10T14:44:11.711737Z","iopub.status.idle":"2026-08-10T14:44:11.725936Z","shell.execute_reply.started":"2026-08-10T14:44:11.711703Z","shell.execute_reply":"2026-08-10T14:44:11.725132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labeled_train = train[train[target_columns].notna().any(axis=1)]\n\nprint(\"Labeled studies:\", len(labeled_train))\nprint(labeled_train[target_columns].notna().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.728442Z","iopub.execute_input":"2026-08-10T14:44:11.729287Z","iopub.status.idle":"2026-08-10T14:44:11.741692Z","shell.execute_reply.started":"2026-08-10T14:44:11.729242Z","shell.execute_reply":"2026-08-10T14:44:11.741003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option(\"display.max_colwidth\", 150)\n\nlabeled_train[\n    [\"StudyInstanceUID\", \"Report\"] + target_columns\n].head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.742634Z","iopub.execute_input":"2026-08-10T14:44:11.742891Z","iopub.status.idle":"2026-08-10T14:44:11.75861Z","shell.execute_reply.started":"2026-08-10T14:44:11.742868Z","shell.execute_reply":"2026-08-10T14:44:11.757812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"study_id = labeled_train.iloc[0][\"StudyInstanceUID\"]\n\nprint(\"Study:\", study_id)\n\nstudy_series = train_series[\n    train_series[\"StudyInstanceUID\"] == study_id\n]\n\nstudy_series","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.75944Z","iopub.execute_input":"2026-08-10T14:44:11.759717Z","iopub.status.idle":"2026-08-10T14:44:11.778567Z","shell.execute_reply.started":"2026-08-10T14:44:11.759696Z","shell.execute_reply":"2026-08-10T14:44:11.777955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"series_id = study_series.iloc[0][\"SeriesInstanceUID\"]\n\nseries_path = os.path.join(\n    competition_path,\n    \"train_series\",\n    series_id\n)\n\nprint(series_path)\nprint(\"Exists:\", os.path.exists(series_path))\n\nif os.path.exists(series_path):\n    files = os.listdir(series_path)\n    print(\"Number of files:\", len(files))\n    print(\"First 10 files:\", files[:10])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.779566Z","iopub.execute_input":"2026-08-10T14:44:11.779836Z","iopub.status.idle":"2026-08-10T14:44:11.790633Z","shell.execute_reply.started":"2026-08-10T14:44:11.779803Z","shell.execute_reply":"2026-08-10T14:44:11.7898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_series_path = os.path.join(competition_path, \"train_series\")\n\nitems = os.listdir(train_series_path)\n\nprint(\"Number of items:\", len(items))\nprint(\"First 20 items:\")\n\nfor item in items[:20]:\n    print(item)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.791499Z","iopub.execute_input":"2026-08-10T14:44:11.792369Z","iopub.status.idle":"2026-08-10T14:44:11.806851Z","shell.execute_reply.started":"2026-08-10T14:44:11.792336Z","shell.execute_reply":"2026-08-10T14:44:11.806133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"study_path = os.path.join(\n    competition_path,\n    \"train_series\",\n    study_id\n)\n\nprint(\"Exists:\", os.path.exists(study_path))\n\nif os.path.exists(study_path):\n    print(\"Number of series:\", len(os.listdir(study_path)))\n    print(\"Series:\")\n    for item in os.listdir(study_path):\n        print(item)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.807717Z","iopub.execute_input":"2026-08-10T14:44:11.808237Z","iopub.status.idle":"2026-08-10T14:44:11.814796Z","shell.execute_reply.started":"2026-08-10T14:44:11.808214Z","shell.execute_reply":"2026-08-10T14:44:11.81392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"series_id = study_series.iloc[0][\"SeriesInstanceUID\"]\n\nseries_path = os.path.join(\n    study_path,\n    series_id\n)\n\nprint(\"Series:\", series_id)\nprint(\"Exists:\", os.path.exists(series_path))\n\nfiles = os.listdir(series_path)\n\nprint(\"Number of files:\", len(files))\nprint(\"First 10 files:\")\n\nfor file in files[:10]:\n    print(file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.815831Z","iopub.execute_input":"2026-08-10T14:44:11.816219Z","iopub.status.idle":"2026-08-10T14:44:11.827845Z","shell.execute_reply.started":"2026-08-10T14:44:11.816196Z","shell.execute_reply":"2026-08-10T14:44:11.826949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.828778Z","iopub.execute_input":"2026-08-10T14:44:11.829071Z","iopub.status.idle":"2026-08-10T14:44:11.835926Z","shell.execute_reply.started":"2026-08-10T14:44:11.829023Z","shell.execute_reply":"2026-08-10T14:44:11.835129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"first_file = os.path.join(series_path, files[0])\n\nds = pydicom.dcmread(first_file)\n\nprint(ds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.837285Z","iopub.execute_input":"2026-08-10T14:44:11.83785Z","iopub.status.idle":"2026-08-10T14:44:11.852489Z","shell.execute_reply.started":"2026-08-10T14:44:11.837828Z","shell.execute_reply":"2026-08-10T14:44:11.851721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nimage = ds.pixel_array\n\nprint(\"Shape:\", image.shape)\nprint(\"Min:\", image.min())\nprint(\"Max:\", image.max())\nprint(\"Mean:\", image.mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.853291Z","iopub.execute_input":"2026-08-10T14:44:11.853497Z","iopub.status.idle":"2026-08-10T14:44:11.860091Z","shell.execute_reply.started":"2026-08-10T14:44:11.853478Z","shell.execute_reply":"2026-08-10T14:44:11.859341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(6, 6))\nplt.imshow(image, cmap=\"gray\")\nplt.axis(\"off\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:11.860799Z","iopub.execute_input":"2026-08-10T14:44:11.86104Z","iopub.status.idle":"2026-08-10T14:44:11.998976Z","shell.execute_reply.started":"2026-08-10T14:44:11.861013Z","shell.execute_reply":"2026-08-10T14:44:11.997769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"slice_info = []\n\nfor file in files:\n    path = os.path.join(series_path, file)\n    dcm = pydicom.dcmread(path, stop_before_pixels=True)\n\n    slice_info.append({\n        \"file\": file,\n        \"instance\": getattr(dcm, \"InstanceNumber\", None),\n        \"slice_location\": getattr(dcm, \"SliceLocation\", None),\n        \"position\": getattr(dcm, \"ImagePositionPatient\", None)\n    })\n\nslice_info = pd.DataFrame(slice_info)\n\nslice_info.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:12.000028Z","iopub.execute_input":"2026-08-10T14:44:12.000421Z","iopub.status.idle":"2026-08-10T14:44:12.059684Z","shell.execute_reply.started":"2026-08-10T14:44:12.000372Z","shell.execute_reply":"2026-08-10T14:44:12.058828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sorted_slices = slice_info.sort_values(\"slice_location\").reset_index(drop=True)\n\nprint(sorted_slices[[\"instance\", \"slice_location\"]].to_string(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:12.060651Z","iopub.execute_input":"2026-08-10T14:44:12.061379Z","iopub.status.idle":"2026-08-10T14:44:12.068224Z","shell.execute_reply.started":"2026-08-10T14:44:12.061354Z","shell.execute_reply":"2026-08-10T14:44:12.067321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_indices = [0, 8, 17, 26, 35]\n\nplt.figure(figsize=(15, 3))\n\nfor i, idx in enumerate(selected_indices):\n    file = sorted_slices.iloc[idx][\"file\"]\n    path = os.path.join(series_path, file)\n\n    dcm = pydicom.dcmread(path)\n    img = dcm.pixel_array\n\n    plt.subplot(1, 5, i + 1)\n    plt.imshow(img, cmap=\"gray\")\n    plt.title(f\"Slice {idx + 1}\")\n    plt.axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:12.069594Z","iopub.execute_input":"2026-08-10T14:44:12.069895Z","iopub.status.idle":"2026-08-10T14:44:12.500325Z","shell.execute_reply.started":"2026-08-10T14:44:12.069864Z","shell.execute_reply":"2026-08-10T14:44:12.499382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labeled_ids = labeled_train[\"StudyInstanceUID\"]\n\nlabeled_series = train_series[\n    train_series[\"StudyInstanceUID\"].isin(labeled_ids)\n]\n\nprint(\"Total series for labeled studies:\", len(labeled_series))\n\nprint(\"\\nSeries per study:\")\nprint(labeled_series.groupby(\"StudyInstanceUID\").size().describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:12.501684Z","iopub.execute_input":"2026-08-10T14:44:12.502115Z","iopub.status.idle":"2026-08-10T14:44:12.513104Z","shell.execute_reply.started":"2026-08-10T14:44:12.502074Z","shell.execute_reply":"2026-08-10T14:44:12.512255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\nAnatomical planes:\")\nprint(labeled_series[\"Anatomical_Plane\"].value_counts())\n\nprint(\"\\nFluid sensitive:\")\nprint(labeled_series[\"Fluid_Sensitive\"].value_counts())\n\nprint(\"\\nFat suppression:\")\nprint(labeled_series[\"Fat_Suppression\"].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:12.51424Z","iopub.execute_input":"2026-08-10T14:44:12.514518Z","iopub.status.idle":"2026-08-10T14:44:12.52457Z","shell.execute_reply.started":"2026-08-10T14:44:12.514486Z","shell.execute_reply":"2026-08-10T14:44:12.523807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"series_summary = (\n    labeled_series\n    .groupby([\"Anatomical_Plane\", \"Fluid_Sensitive\", \"Fat_Suppression\"])\n    .size()\n    .reset_index(name=\"count\")\n)\n\nseries_summary","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:12.525503Z","iopub.execute_input":"2026-08-10T14:44:12.525812Z","iopub.status.idle":"2026-08-10T14:44:12.544214Z","shell.execute_reply.started":"2026-08-10T14:44:12.525779Z","shell.execute_reply":"2026-08-10T14:44:12.543616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"slice_counts = []\n\nfor _, row in labeled_series.iterrows():\n    study = row[\"StudyInstanceUID\"]\n    series = row[\"SeriesInstanceUID\"]\n\n    path = os.path.join(\n        competition_path,\n        \"train_series\",\n        study,\n        series\n    )\n\n    slice_counts.append({\n        \"StudyInstanceUID\": study,\n        \"SeriesInstanceUID\": series,\n        \"Plane\": row[\"Anatomical_Plane\"],\n        \"Fluid_Sensitive\": row[\"Fluid_Sensitive\"],\n        \"Slices\": len(os.listdir(path))\n    })\n\nslice_counts = pd.DataFrame(slice_counts)\n\nslice_counts.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:12.54495Z","iopub.execute_input":"2026-08-10T14:44:12.545221Z","iopub.status.idle":"2026-08-10T14:44:12.842621Z","shell.execute_reply.started":"2026-08-10T14:44:12.545199Z","shell.execute_reply":"2026-08-10T14:44:12.84191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"series_presence = (\n    labeled_series\n    .groupby([\"StudyInstanceUID\", \"Anatomical_Plane\", \"Fluid_Sensitive\"])\n    .size()\n    .unstack(fill_value=0)\n)\n\nseries_presence","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:12.847678Z","iopub.execute_input":"2026-08-10T14:44:12.848003Z","iopub.status.idle":"2026-08-10T14:44:12.861761Z","shell.execute_reply.started":"2026-08-10T14:44:12.847979Z","shell.execute_reply":"2026-08-10T14:44:12.861118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"series_presence_reset = series_presence.reset_index()\n\nprint(series_presence_reset.head(20))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:12.862711Z","iopub.execute_input":"2026-08-10T14:44:12.86294Z","iopub.status.idle":"2026-08-10T14:44:12.870488Z","shell.execute_reply.started":"2026-08-10T14:44:12.862918Z","shell.execute_reply":"2026-08-10T14:44:12.869798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sagittal_fs = labeled_series[\n    (labeled_series[\"Anatomical_Plane\"] == \"Sagittal\") &\n    (labeled_series[\"Fluid_Sensitive\"] == 1)\n]\n\nprint(\"Labeled studies:\", labeled_train[\"StudyInstanceUID\"].nunique())\nprint(\"Studies with sagittal fluid-sensitive series:\",\n      sagittal_fs[\"StudyInstanceUID\"].nunique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:12.871458Z","iopub.execute_input":"2026-08-10T14:44:12.872192Z","iopub.status.idle":"2026-08-10T14:44:12.884756Z","shell.execute_reply.started":"2026-08-10T14:44:12.872162Z","shell.execute_reply":"2026-08-10T14:44:12.884132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_labeled_ids = set(labeled_train[\"StudyInstanceUID\"])\nsagittal_fs_ids = set(sagittal_fs[\"StudyInstanceUID\"])\n\nmissing_sagittal_fs = all_labeled_ids - sagittal_fs_ids\n\nprint(\"Missing studies:\", len(missing_sagittal_fs))\n\nfor study in missing_sagittal_fs:\n    print(study)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:12.885899Z","iopub.execute_input":"2026-08-10T14:44:12.886624Z","iopub.status.idle":"2026-08-10T14:44:12.897271Z","shell.execute_reply.started":"2026-08-10T14:44:12.886588Z","shell.execute_reply":"2026-08-10T14:44:12.896526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_series = labeled_series[\n    labeled_series[\"StudyInstanceUID\"].isin(missing_sagittal_fs)\n]\n\nmissing_series[\n    [\"StudyInstanceUID\", \"SeriesInstanceUID\",\n     \"Anatomical_Plane\", \"Fluid_Sensitive\", \"Fat_Suppression\"]\n].sort_values([\"StudyInstanceUID\", \"Anatomical_Plane\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:12.89836Z","iopub.execute_input":"2026-08-10T14:44:12.899254Z","iopub.status.idle":"2026-08-10T14:44:12.915707Z","shell.execute_reply.started":"2026-08-10T14:44:12.899222Z","shell.execute_reply":"2026-08-10T14:44:12.91513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sagittal_counts = (\n    sagittal_fs\n    .groupby(\"StudyInstanceUID\")\n    .size()\n)\n\nprint(sagittal_counts.value_counts().sort_index())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:12.916683Z","iopub.execute_input":"2026-08-10T14:44:12.916996Z","iopub.status.idle":"2026-08-10T14:44:12.928882Z","shell.execute_reply.started":"2026-08-10T14:44:12.91695Z","shell.execute_reply":"2026-08-10T14:44:12.928137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"duplicate_sagittal_fs = sagittal_fs[\n    sagittal_fs[\"StudyInstanceUID\"].isin(\n        sagittal_counts[sagittal_counts > 1].index\n    )\n]\n\nduplicate_sagittal_fs[\n    [\"StudyInstanceUID\", \"SeriesInstanceUID\"]\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:12.929744Z","iopub.execute_input":"2026-08-10T14:44:12.930007Z","iopub.status.idle":"2026-08-10T14:44:12.944839Z","shell.execute_reply.started":"2026-08-10T14:44:12.929987Z","shell.execute_reply":"2026-08-10T14:44:12.94397Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"duplicate_info = slice_counts[\n    slice_counts[\"SeriesInstanceUID\"].isin(\n        duplicate_sagittal_fs[\"SeriesInstanceUID\"]\n    )\n]\n\nduplicate_info[\n    [\"StudyInstanceUID\", \"SeriesInstanceUID\", \"Plane\",\n     \"Fluid_Sensitive\", \"Slices\"]\n].sort_values([\"StudyInstanceUID\", \"Slices\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:12.945872Z","iopub.execute_input":"2026-08-10T14:44:12.946362Z","iopub.status.idle":"2026-08-10T14:44:12.965557Z","shell.execute_reply.started":"2026-08-10T14:44:12.946325Z","shell.execute_reply":"2026-08-10T14:44:12.964914Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"series_metadata = []\n\nfor _, row in duplicate_info.iterrows():\n    study = row[\"StudyInstanceUID\"]\n    series = row[\"SeriesInstanceUID\"]\n\n    path = os.path.join(\n        competition_path,\n        \"train_series\",\n        study,\n        series\n    )\n\n    # Read only the first DICOM file; don't load pixel data\n    first_dcm_file = os.listdir(path)[0]\n    dcm_path = os.path.join(path, first_dcm_file)\n\n    dcm = pydicom.dcmread(dcm_path, stop_before_pixels=True)\n\n    series_metadata.append({\n        \"StudyInstanceUID\": study,\n        \"SeriesInstanceUID\": series,\n        \"SeriesDescription\": getattr(dcm, \"SeriesDescription\", \"\"),\n        \"SequenceName\": getattr(dcm, \"SequenceName\", \"\"),\n        \"ScanningSequence\": getattr(dcm, \"ScanningSequence\", \"\"),\n        \"Rows\": getattr(dcm, \"Rows\", None),\n        \"Columns\": getattr(dcm, \"Columns\", None),\n        \"SliceThickness\": getattr(dcm, \"SliceThickness\", None),\n        \"SpacingBetweenSlices\": getattr(dcm, \"SpacingBetweenSlices\", None),\n    })\n\nseries_metadata = pd.DataFrame(series_metadata)\n\nseries_metadata","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:12.966348Z","iopub.execute_input":"2026-08-10T14:44:12.966642Z","iopub.status.idle":"2026-08-10T14:44:13.018666Z","shell.execute_reply.started":"2026-08-10T14:44:12.966619Z","shell.execute_reply":"2026-08-10T14:44:13.018009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_series[[\"Anatomical_Plane\", \"Fluid_Sensitive\", \"Fat_Suppression\"]].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:13.01946Z","iopub.execute_input":"2026-08-10T14:44:13.019818Z","iopub.status.idle":"2026-08-10T14:44:13.028258Z","shell.execute_reply.started":"2026-08-10T14:44:13.019785Z","shell.execute_reply":"2026-08-10T14:44:13.027447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_series","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:13.029202Z","iopub.execute_input":"2026-08-10T14:44:13.029794Z","iopub.status.idle":"2026-08-10T14:44:13.042413Z","shell.execute_reply.started":"2026-08-10T14:44:13.029771Z","shell.execute_reply":"2026-08-10T14:44:13.041739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_metadata = []\n\nfor _, row in test_series.iterrows():\n    study = row[\"StudyInstanceUID\"]\n    series = row[\"SeriesInstanceUID\"]\n\n    path = os.path.join(\n        competition_path,\n        \"test_series\",\n        study,\n        series\n    )\n\n    first_file = os.listdir(path)[0]\n    dcm = pydicom.dcmread(\n        os.path.join(path, first_file),\n        stop_before_pixels=True\n    )\n\n    test_metadata.append({\n        \"StudyInstanceUID\": study,\n        \"SeriesInstanceUID\": series,\n        \"Plane\": row[\"Anatomical_Plane\"],\n        \"Fluid_Sensitive\": row[\"Fluid_Sensitive\"],\n        \"SeriesDescription\": getattr(dcm, \"SeriesDescription\", \"\")\n    })\n\ntest_metadata = pd.DataFrame(test_metadata)\n\ntest_metadata","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:13.043315Z","iopub.execute_input":"2026-08-10T14:44:13.043707Z","iopub.status.idle":"2026-08-10T14:44:13.090776Z","shell.execute_reply.started":"2026-08-10T14:44:13.04367Z","shell.execute_reply":"2026-08-10T14:44:13.089999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_slice_counts = []\n\nfor _, row in test_series.iterrows():\n    study = row[\"StudyInstanceUID\"]\n    series = row[\"SeriesInstanceUID\"]\n\n    path = os.path.join(\n        competition_path,\n        \"test_series\",\n        study,\n        series\n    )\n\n    test_slice_counts.append({\n        \"StudyInstanceUID\": study,\n        \"SeriesInstanceUID\": series,\n        \"Plane\": row[\"Anatomical_Plane\"],\n        \"Fluid_Sensitive\": row[\"Fluid_Sensitive\"],\n        \"SeriesDescription\": test_metadata.loc[\n            test_metadata[\"SeriesInstanceUID\"] == series,\n            \"SeriesDescription\"\n        ].iloc[0],\n        \"Slices\": len(os.listdir(path))\n    })\n\ntest_slice_counts = pd.DataFrame(test_slice_counts)\n\ntest_slice_counts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:13.091684Z","iopub.execute_input":"2026-08-10T14:44:13.091957Z","iopub.status.idle":"2026-08-10T14:44:13.117642Z","shell.execute_reply.started":"2026-08-10T14:44:13.091927Z","shell.execute_reply":"2026-08-10T14:44:13.117035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fs_series = labeled_series[labeled_series[\"Fluid_Sensitive\"] == 1].copy()\n\nfs_metadata = []\n\nfor _, row in fs_series.iterrows():\n    study = row[\"StudyInstanceUID\"]\n    series = row[\"SeriesInstanceUID\"]\n\n    path = os.path.join(\n        competition_path,\n        \"train_series\",\n        study,\n        series\n    )\n\n    first_file = os.listdir(path)[0]\n\n    dcm = pydicom.dcmread(\n        os.path.join(path, first_file),\n        stop_before_pixels=True\n    )\n\n    fs_metadata.append({\n        \"StudyInstanceUID\": study,\n        \"SeriesInstanceUID\": series,\n        \"Plane\": row[\"Anatomical_Plane\"],\n        \"SeriesDescription\": getattr(dcm, \"SeriesDescription\", \"\")\n    })\n\nfs_metadata = pd.DataFrame(fs_metadata)\n\nprint(fs_metadata.shape)\nfs_metadata.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:13.118493Z","iopub.execute_input":"2026-08-10T14:44:13.118771Z","iopub.status.idle":"2026-08-10T14:44:13.491028Z","shell.execute_reply.started":"2026-08-10T14:44:13.118737Z","shell.execute_reply":"2026-08-10T14:44:13.490236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fs_metadata[\"SeriesDescription\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:13.492009Z","iopub.execute_input":"2026-08-10T14:44:13.492367Z","iopub.status.idle":"2026-08-10T14:44:13.499465Z","shell.execute_reply.started":"2026-08-10T14:44:13.492334Z","shell.execute_reply":"2026-08-10T14:44:13.498666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for plane in [\"Sagittal\", \"Coronal\", \"Axial\"]:\n    print(f\"\\n===== {plane} =====\")\n\n    descriptions = (\n        fs_metadata.loc[\n            fs_metadata[\"Plane\"] == plane,\n            \"SeriesDescription\"\n        ]\n        .fillna(\"\")\n        .unique()\n    )\n\n    for desc in sorted(descriptions):\n        print(repr(desc))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:13.50054Z","iopub.execute_input":"2026-08-10T14:44:13.500832Z","iopub.status.idle":"2026-08-10T14:44:13.513655Z","shell.execute_reply.started":"2026-08-10T14:44:13.500778Z","shell.execute_reply":"2026-08-10T14:44:13.512932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def select_series(series_df):\n    \"\"\"\n    Select one MRI series for a study.\n\n    Priority:\n    1. Sagittal fluid-sensitive\n    2. Coronal fluid-sensitive\n    3. Axial fluid-sensitive\n    4. Sagittal standard\n    5. Coronal standard\n    6. Axial standard\n    \"\"\"\n\n    plane_priority = {\n        \"Sagittal\": 0,\n        \"Coronal\": 1,\n        \"Axial\": 2\n    }\n\n    candidates = series_df.copy()\n\n    # Fluid-sensitive first\n    candidates[\"fluid_priority\"] = (\n        candidates[\"Fluid_Sensitive\"] == 0\n    ).astype(int)\n\n    # Anatomical plane priority\n    candidates[\"plane_priority\"] = (\n        candidates[\"Anatomical_Plane\"]\n        .map(plane_priority)\n        .fillna(99)\n    )\n\n    # Sort deterministically\n    candidates = candidates.sort_values(\n        [\"fluid_priority\", \"plane_priority\", \"SeriesInstanceUID\"]\n    )\n\n    return candidates.iloc[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:13.514557Z","iopub.execute_input":"2026-08-10T14:44:13.515268Z","iopub.status.idle":"2026-08-10T14:44:13.52234Z","shell.execute_reply.started":"2026-08-10T14:44:13.515245Z","shell.execute_reply":"2026-08-10T14:44:13.521634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected = select_series(study_series)\n\nprint(\"Selected series:\")\nprint(selected)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:13.523457Z","iopub.execute_input":"2026-08-10T14:44:13.524239Z","iopub.status.idle":"2026-08-10T14:44:13.536784Z","shell.execute_reply.started":"2026-08-10T14:44:13.524209Z","shell.execute_reply":"2026-08-10T14:44:13.536118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_series = (\n    labeled_series\n    .groupby(\"StudyInstanceUID\", group_keys=False)\n    .apply(select_series)\n    .reset_index(drop=True)\n)\n\nprint(\"Selected studies:\", selected_series[\"StudyInstanceUID\"].nunique())\nprint(\"\\nSelected planes:\")\nprint(selected_series[\"Anatomical_Plane\"].value_counts())\n\nprint(\"\\nSelected fluid-sensitive:\")\nprint(selected_series[\"Fluid_Sensitive\"].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:13.537748Z","iopub.execute_input":"2026-08-10T14:44:13.53864Z","iopub.status.idle":"2026-08-10T14:44:13.665043Z","shell.execute_reply.started":"2026-08-10T14:44:13.538615Z","shell.execute_reply":"2026-08-10T14:44:13.664258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_series_slices(study_uid, series_uid):\n    series_path = os.path.join(\n        competition_path,\n        \"train_series\",\n        study_uid,\n        series_uid\n    )\n\n    slice_data = []\n\n    for filename in os.listdir(series_path):\n        path = os.path.join(series_path, filename)\n\n        dcm = pydicom.dcmread(path)\n\n        position = np.array(dcm.ImagePositionPatient, dtype=float)\n\n        slice_data.append({\n            \"filename\": filename,\n            \"position\": position,\n            \"image\": dcm.pixel_array\n        })\n\n    # Sort using the first coordinate of ImagePositionPatient\n    slice_data.sort(key=lambda x: x[\"position\"][0])\n\n    return slice_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:13.66603Z","iopub.execute_input":"2026-08-10T14:44:13.666373Z","iopub.status.idle":"2026-08-10T14:44:13.671847Z","shell.execute_reply.started":"2026-08-10T14:44:13.66634Z","shell.execute_reply":"2026-08-10T14:44:13.671148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"row = selected_series.iloc[0]\n\nslices = load_series_slices(\n    row[\"StudyInstanceUID\"],\n    row[\"SeriesInstanceUID\"]\n)\n\nprint(\"Number of slices:\", len(slices))\nprint(\"First slice shape:\", slices[0][\"image\"].shape)\nprint(\"Last slice shape:\", slices[-1][\"image\"].shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:13.672716Z","iopub.execute_input":"2026-08-10T14:44:13.672989Z","iopub.status.idle":"2026-08-10T14:44:13.744569Z","shell.execute_reply.started":"2026-08-10T14:44:13.672969Z","shell.execute_reply":"2026-08-10T14:44:13.743752Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nindices = [0, 8, 17, 26, 35]\n\nplt.figure(figsize=(15, 3))\n\nfor i, idx in enumerate(indices):\n    plt.subplot(1, 5, i + 1)\n    plt.imshow(slices[idx][\"image\"], cmap=\"gray\")\n    plt.title(f\"Slice {idx}\")\n    plt.axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:13.745467Z","iopub.execute_input":"2026-08-10T14:44:13.74587Z","iopub.status.idle":"2026-08-10T14:44:14.120773Z","shell.execute_reply.started":"2026-08-10T14:44:13.745846Z","shell.execute_reply":"2026-08-10T14:44:14.120013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def normalize_slice(image):\n    image = image.astype(np.float32)\n\n    low = np.percentile(image, 1)\n    high = np.percentile(image, 99)\n\n    image = np.clip(image, low, high)\n    image = (image - low) / (high - low + 1e-8)\n\n    return image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:14.122239Z","iopub.execute_input":"2026-08-10T14:44:14.122655Z","iopub.status.idle":"2026-08-10T14:44:14.128251Z","shell.execute_reply.started":"2026-08-10T14:44:14.122621Z","shell.execute_reply":"2026-08-10T14:44:14.127362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"normalized = normalize_slice(slices[17][\"image\"])\n\nprint(\"Min:\", normalized.min())\nprint(\"Max:\", normalized.max())\nprint(\"Mean:\", normalized.mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:14.129119Z","iopub.execute_input":"2026-08-10T14:44:14.129508Z","iopub.status.idle":"2026-08-10T14:44:14.144123Z","shell.execute_reply.started":"2026-08-10T14:44:14.129466Z","shell.execute_reply":"2026-08-10T14:44:14.143409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def sample_slices(slices, num_slices=16):\n    total = len(slices)\n\n    if total <= num_slices:\n        indices = np.linspace(0, total - 1, num_slices).round().astype(int)\n    else:\n        indices = np.linspace(0, total - 1, num_slices).round().astype(int)\n\n    return [slices[i][\"image\"] for i in indices]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:14.14519Z","iopub.execute_input":"2026-08-10T14:44:14.145479Z","iopub.status.idle":"2026-08-10T14:44:14.153191Z","shell.execute_reply.started":"2026-08-10T14:44:14.145447Z","shell.execute_reply":"2026-08-10T14:44:14.152528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\n\ndef preprocess_slices(slices, num_slices=16, image_size=224):\n    selected = sample_slices(slices, num_slices)\n\n    processed = []\n\n    for image in selected:\n        image = normalize_slice(image)\n\n        image = Image.fromarray(\n            (image * 255).astype(np.uint8)\n        )\n\n        image = image.resize(\n            (image_size, image_size),\n            Image.Resampling.BILINEAR\n        )\n\n        image = np.asarray(image, dtype=np.float32) / 255.0\n\n        processed.append(image)\n\n    return np.stack(processed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:14.153908Z","iopub.execute_input":"2026-08-10T14:44:14.154255Z","iopub.status.idle":"2026-08-10T14:44:14.165273Z","shell.execute_reply.started":"2026-08-10T14:44:14.154231Z","shell.execute_reply":"2026-08-10T14:44:14.16459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"volume = preprocess_slices(slices)\n\nprint(\"Shape:\", volume.shape)\nprint(\"Min:\", volume.min())\nprint(\"Max:\", volume.max())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:14.165955Z","iopub.execute_input":"2026-08-10T14:44:14.16624Z","iopub.status.idle":"2026-08-10T14:44:14.214618Z","shell.execute_reply.started":"2026-08-10T14:44:14.166207Z","shell.execute_reply":"2026-08-10T14:44:14.213984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 8, figsize=(16, 4))\n\nfor i, ax in enumerate(axes.flat):\n    ax.imshow(volume[i], cmap=\"gray\")\n    ax.set_title(f\"{i + 1}\")\n    ax.axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:14.215409Z","iopub.execute_input":"2026-08-10T14:44:14.215588Z","iopub.status.idle":"2026-08-10T14:44:15.219844Z","shell.execute_reply.started":"2026-08-10T14:44:14.215569Z","shell.execute_reply":"2026-08-10T14:44:15.218936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_columns = [\n    \"ACL\",\n    \"MCL\",\n    \"Medial Meniscus\",\n    \"Lateral Meniscus\",\n    \"Medial OA\",\n    \"Lateral OA\",\n    \"PF OA\",\n    \"Effusion\",\n    \"Synovitis\",\n    \"Baker's\",\n    \"Contusion\",\n    \"Fracture\"\n]\n\nlabels = train[\n    [\"StudyInstanceUID\"] + target_columns\n].copy()\n\nlabels = labels.dropna(subset=target_columns)\n\nprint(\"Labeled studies:\", len(labels))\nprint(\"Label shape:\", labels[target_columns].shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:15.221349Z","iopub.execute_input":"2026-08-10T14:44:15.221606Z","iopub.status.idle":"2026-08-10T14:44:15.231524Z","shell.execute_reply.started":"2026-08-10T14:44:15.221583Z","shell.execute_reply":"2026-08-10T14:44:15.23077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_ids, val_ids = train_test_split(\n    labels[\"StudyInstanceUID\"].values,\n    test_size=0.20,\n    random_state=42,\n    shuffle=True\n)\n\ntrain_labels = labels[\n    labels[\"StudyInstanceUID\"].isin(train_ids)\n].reset_index(drop=True)\n\nval_labels = labels[\n    labels[\"StudyInstanceUID\"].isin(val_ids)\n].reset_index(drop=True)\n\nprint(\"Training studies:\", len(train_labels))\nprint(\"Validation studies:\", len(val_labels))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:15.23254Z","iopub.execute_input":"2026-08-10T14:44:15.232828Z","iopub.status.idle":"2026-08-10T14:44:15.243487Z","shell.execute_reply.started":"2026-08-10T14:44:15.232793Z","shell.execute_reply":"2026-08-10T14:44:15.242615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"TRAIN positives:\")\nprint(train_labels[target_columns].sum())\n\nprint(\"\\nVALIDATION positives:\")\nprint(val_labels[target_columns].sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:15.244367Z","iopub.execute_input":"2026-08-10T14:44:15.24464Z","iopub.status.idle":"2026-08-10T14:44:15.261838Z","shell.execute_reply.started":"2026-08-10T14:44:15.244608Z","shell.execute_reply":"2026-08-10T14:44:15.261114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Find a good 80/20 split by repeatedly trying random study splits.\n# We select the split whose validation label counts are closest\n# to the expected 20% of the complete labeled dataset.\n\ny = labels[target_columns].values.astype(int)\n\nn_total = len(labels)\nn_val = 12\n\ntotal_counts = y.sum(axis=0)\nexpected_val = total_counts * n_val / n_total\n\nbest_score = np.inf\nbest_val_indices = None\n\nrng = np.random.default_rng(42)\n\nfor _ in range(10000):\n\n    val_indices = rng.choice(\n        n_total,\n        size=n_val,\n        replace=False\n    )\n\n    val_counts = y[val_indices].sum(axis=0)\n\n    # Penalize zero-positive validation classes heavily\n    zero_penalty = np.sum(val_counts == 0) * 100\n\n    # Difference from expected number of positives\n    distribution_error = np.sum(\n        (val_counts - expected_val) ** 2\n    )\n\n    score = zero_penalty + distribution_error\n\n    if score < best_score:\n        best_score = score\n        best_val_indices = val_indices.copy()\n\n\n# Create final split\nval_indices = np.sort(best_val_indices)\n\nall_indices = np.arange(n_total)\n\ntrain_indices = np.setdiff1d(\n    all_indices,\n    val_indices\n)\n\ntrain_labels = labels.iloc[train_indices].reset_index(drop=True)\nval_labels = labels.iloc[val_indices].reset_index(drop=True)\n\nprint(\"Training studies:\", len(train_labels))\nprint(\"Validation studies:\", len(val_labels))\n\nprint(\"\\nTotal positives:\")\nprint(labels[target_columns].sum())\n\nprint(\"\\nTRAIN positives:\")\nprint(train_labels[target_columns].sum())\n\nprint(\"\\nVALIDATION positives:\")\nprint(val_labels[target_columns].sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:15.26282Z","iopub.execute_input":"2026-08-10T14:44:15.263156Z","iopub.status.idle":"2026-08-10T14:44:15.602678Z","shell.execute_reply.started":"2026-08-10T14:44:15.263124Z","shell.execute_reply":"2026-08-10T14:44:15.601824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\n\nprint(\"PyTorch:\", torch.__version__)\nprint(\"CUDA available:\", torch.cuda.is_available())\n\nif torch.cuda.is_available():\n    print(\"GPU:\", torch.cuda.get_device_name(0))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:15.603665Z","iopub.execute_input":"2026-08-10T14:44:15.603961Z","iopub.status.idle":"2026-08-10T14:44:15.608826Z","shell.execute_reply.started":"2026-08-10T14:44:15.603929Z","shell.execute_reply":"2026-08-10T14:44:15.607918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch.utils.data import Dataset\n\nclass KneeMRIDataset(Dataset):\n\n    def __init__(self, labels_df, selected_series_df):\n        self.labels_df = labels_df.reset_index(drop=True)\n\n        self.series_map = (\n            selected_series_df\n            .set_index(\"StudyInstanceUID\")\n        )\n\n    def __len__(self):\n        return len(self.labels_df)\n\n    def __getitem__(self, idx):\n\n        row = self.labels_df.iloc[idx]\n\n        study_uid = row[\"StudyInstanceUID\"]\n\n        series_info = self.series_map.loc[study_uid]\n\n        series_uid = series_info[\"SeriesInstanceUID\"]\n\n        # Load DICOM slices\n        slices = load_series_slices(\n            study_uid,\n            series_uid\n        )\n\n        # Preprocess → (16, 224, 224)\n        volume = preprocess_slices(\n            slices,\n            num_slices=16,\n            image_size=224\n        )\n\n        # Convert to tensor\n        # Current shape: (16, 224, 224)\n        image = torch.tensor(\n            volume,\n            dtype=torch.float32\n        )\n\n        # Labels → (12,)\n        target = torch.tensor(\n            row[target_columns].values.astype(np.float32),\n            dtype=torch.float32\n        )\n\n        return image, target","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:15.609674Z","iopub.execute_input":"2026-08-10T14:44:15.610004Z","iopub.status.idle":"2026-08-10T14:44:15.621256Z","shell.execute_reply.started":"2026-08-10T14:44:15.609969Z","shell.execute_reply":"2026-08-10T14:44:15.620495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = KneeMRIDataset(\n    train_labels,\n    selected_series\n)\n\nval_dataset = KneeMRIDataset(\n    val_labels,\n    selected_series\n)\n\nprint(\"Train dataset:\", len(train_dataset))\nprint(\"Validation dataset:\", len(val_dataset))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:15.622205Z","iopub.execute_input":"2026-08-10T14:44:15.622418Z","iopub.status.idle":"2026-08-10T14:44:15.63788Z","shell.execute_reply.started":"2026-08-10T14:44:15.622397Z","shell.execute_reply":"2026-08-10T14:44:15.637081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image, target = train_dataset[0]\n\nprint(\"Image shape:\", image.shape)\nprint(\"Image dtype:\", image.dtype)\nprint(\"Image min:\", image.min().item())\nprint(\"Image max:\", image.max().item())\n\nprint(\"\\nTarget shape:\", target.shape)\nprint(\"Target:\", target)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:15.638741Z","iopub.execute_input":"2026-08-10T14:44:15.639708Z","iopub.status.idle":"2026-08-10T14:44:15.914001Z","shell.execute_reply.started":"2026-08-10T14:44:15.639671Z","shell.execute_reply":"2026-08-10T14:44:15.913239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch.utils.data import DataLoader\n\ntrain_loader = DataLoader(\n    train_dataset,\n    batch_size=4,\n    shuffle=True,\n    num_workers=2,\n    pin_memory=True\n)\n\nval_loader = DataLoader(\n    val_dataset,\n    batch_size=4,\n    shuffle=False,\n    num_workers=2,\n    pin_memory=True\n)\n\nprint(\"Train batches:\", len(train_loader))\nprint(\"Validation batches:\", len(val_loader))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:15.914942Z","iopub.execute_input":"2026-08-10T14:44:15.915236Z","iopub.status.idle":"2026-08-10T14:44:15.921458Z","shell.execute_reply.started":"2026-08-10T14:44:15.915211Z","shell.execute_reply":"2026-08-10T14:44:15.920484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"images, targets = next(iter(train_loader))\n\nprint(\"Images shape:\", images.shape)\nprint(\"Images dtype:\", images.dtype)\n\nprint(\"Targets shape:\", targets.shape)\nprint(\"Targets dtype:\", targets.dtype)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:15.922169Z","iopub.execute_input":"2026-08-10T14:44:15.92241Z","iopub.status.idle":"2026-08-10T14:44:18.491379Z","shell.execute_reply.started":"2026-08-10T14:44:15.922381Z","shell.execute_reply":"2026-08-10T14:44:18.490436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def make_25d(images):\n    \"\"\"\n    Convert:\n        (B, 16, H, W)\n    into:\n        (B, 14, 3, H, W)\n    \"\"\"\n\n    return torch.stack(\n        [\n            images[:, i:i+3]\n            for i in range(images.shape[1] - 2)\n        ],\n        dim=1\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:18.493011Z","iopub.execute_input":"2026-08-10T14:44:18.49338Z","iopub.status.idle":"2026-08-10T14:44:18.49877Z","shell.execute_reply.started":"2026-08-10T14:44:18.493345Z","shell.execute_reply":"2026-08-10T14:44:18.49789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"images_25d = make_25d(images)\n\nprint(\"Original:\", images.shape)\nprint(\"2.5D:\", images_25d.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:18.499825Z","iopub.execute_input":"2026-08-10T14:44:18.500511Z","iopub.status.idle":"2026-08-10T14:44:18.531238Z","shell.execute_reply.started":"2026-08-10T14:44:18.500471Z","shell.execute_reply":"2026-08-10T14:44:18.530375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torchvision\n\nprint(\"Torchvision:\", torchvision.__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:18.532283Z","iopub.execute_input":"2026-08-10T14:44:18.532563Z","iopub.status.idle":"2026-08-10T14:44:18.537647Z","shell.execute_reply.started":"2026-08-10T14:44:18.53253Z","shell.execute_reply":"2026-08-10T14:44:18.536801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchvision.models import resnet18, ResNet18_Weights\n\nweights = ResNet18_Weights.DEFAULT\n\nmodel = resnet18(weights=weights)\n\n# ResNet normally outputs 1000 ImageNet classes.\n# Replace the final layer with our 12 abnormalities.\nmodel.fc = torch.nn.Linear(\n    model.fc.in_features,\n    len(target_columns)\n)\n\nprint(model.fc)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:18.53862Z","iopub.execute_input":"2026-08-10T14:44:18.538829Z","iopub.status.idle":"2026-08-10T14:44:18.739638Z","shell.execute_reply.started":"2026-08-10T14:44:18.538808Z","shell.execute_reply":"2026-08-10T14:44:18.738857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nmodel = model.to(device)\n\nprint(\"Device:\", device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:18.740793Z","iopub.execute_input":"2026-08-10T14:44:18.741088Z","iopub.status.idle":"2026-08-10T14:44:18.76187Z","shell.execute_reply.started":"2026-08-10T14:44:18.741043Z","shell.execute_reply":"2026-08-10T14:44:18.761293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"images_gpu = images.to(device, non_blocking=True)\n\nimages_25d = make_25d(images_gpu)\n\nB, N, C, H, W = images_25d.shape\n\n# Flatten the study and 2.5D-group dimensions\ncnn_input = images_25d.reshape(\n    B * N,\n    C,\n    H,\n    W\n)\n\nprint(\"CNN input:\", cnn_input.shape)\n\nwith torch.no_grad():\n    output = model(cnn_input)\n\nprint(\"CNN output:\", output.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:18.762713Z","iopub.execute_input":"2026-08-10T14:44:18.762903Z","iopub.status.idle":"2026-08-10T14:44:18.777597Z","shell.execute_reply.started":"2026-08-10T14:44:18.762884Z","shell.execute_reply":"2026-08-10T14:44:18.776958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"study_output = output.reshape(\n    B,\n    N,\n    len(target_columns)\n)\n\nstudy_output = study_output.mean(dim=1)\n\nprint(\"Study-level output:\", study_output.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:18.778444Z","iopub.execute_input":"2026-08-10T14:44:18.778825Z","iopub.status.idle":"2026-08-10T14:44:18.783881Z","shell.execute_reply.started":"2026-08-10T14:44:18.778789Z","shell.execute_reply":"2026-08-10T14:44:18.783049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"criterion = torch.nn.BCEWithLogitsLoss(\n    pos_weight=pos_weight\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:18.784862Z","iopub.execute_input":"2026-08-10T14:44:18.785239Z","iopub.status.idle":"2026-08-10T14:44:18.794255Z","shell.execute_reply.started":"2026-08-10T14:44:18.785206Z","shell.execute_reply":"2026-08-10T14:44:18.793633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"loss = criterion(\n    study_output,\n    targets.to(device)\n)\n\nprint(\"Loss:\", loss.item())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:18.795132Z","iopub.execute_input":"2026-08-10T14:44:18.795404Z","iopub.status.idle":"2026-08-10T14:44:18.83773Z","shell.execute_reply.started":"2026-08-10T14:44:18.795373Z","shell.execute_reply":"2026-08-10T14:44:18.837128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pos_counts = train_labels[target_columns].sum().values\nneg_counts = len(train_labels) - pos_counts\n\npos_weight = neg_counts / (pos_counts + 1e-8)\n\npos_weight = torch.tensor(\n    pos_weight,\n    dtype=torch.float32,\n    device=device\n)\n\nprint(pd.Series(\n    pos_weight.detach().cpu().numpy(),\n    index=target_columns\n))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:44:18.838606Z","iopub.execute_input":"2026-08-10T14:44:18.839344Z","iopub.status.idle":"2026-08-10T14:44:18.848432Z","shell.execute_reply.started":"2026-08-10T14:44:18.839309Z","shell.execute_reply":"2026-08-10T14:44:18.847703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Freeze the pretrained ResNet backbone\nfor param in model.parameters():\n    param.requires_grad = False\n\n# Only train the final classification layer\nfor param in model.fc.parameters():\n    param.requires_grad = True\n\n# Weighted multilabel loss\ncriterion = torch.nn.BCEWithLogitsLoss(\n    pos_weight=pos_weight\n)\n\n# Optimizer\noptimizer = torch.optim.AdamW(\n    model.fc.parameters(),\n    lr=1e-3,\n    weight_decay=1e-4\n)\n\nprint(\"Optimizer ready\")\nprint(\"Trainable parameters:\",\n      sum(p.numel() for p in model.parameters() if p.requires_grad))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:45:13.863136Z","iopub.execute_input":"2026-08-10T14:45:13.863629Z","iopub.status.idle":"2026-08-10T14:45:13.871868Z","shell.execute_reply.started":"2026-08-10T14:45:13.86359Z","shell.execute_reply":"2026-08-10T14:45:13.87097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.train()\n\nimages, targets = next(iter(train_loader))\n\nimages = images.to(device, non_blocking=True)\ntargets = targets.to(device, non_blocking=True)\n\nimages_25d = make_25d(images)\n\nB, N, C, H, W = images_25d.shape\n\ncnn_input = images_25d.reshape(\n    B * N,\n    C,\n    H,\n    W\n)\n\noptimizer.zero_grad()\n\noutput = model(cnn_input)\n\nstudy_output = output.reshape(\n    B,\n    N,\n    len(target_columns)\n).mean(dim=1)\n\nloss = criterion(\n    study_output,\n    targets\n)\n\nloss.backward()\noptimizer.step()\n\nprint(\"Training batch loss:\", loss.item())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:46:34.901853Z","iopub.execute_input":"2026-08-10T14:46:34.902704Z","iopub.status.idle":"2026-08-10T14:46:36.937391Z","shell.execute_reply.started":"2026-08-10T14:46:34.902672Z","shell.execute_reply":"2026-08-10T14:46:36.936559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_val_loss = float(\"inf\")\nnum_epochs = 10\n\nfor epoch in range(num_epochs):\n\n    # =========================\n    # TRAIN\n    # =========================\n    model.train()\n\n    train_loss = 0.0\n\n    for images, targets in train_loader:\n\n        images = images.to(device, non_blocking=True)\n        targets = targets.to(device, non_blocking=True)\n\n        images_25d = make_25d(images)\n\n        B, N, C, H, W = images_25d.shape\n\n        cnn_input = images_25d.reshape(\n            B * N,\n            C,\n            H,\n            W\n        )\n\n        optimizer.zero_grad()\n\n        output = model(cnn_input)\n\n        study_output = output.reshape(\n            B,\n            N,\n            len(target_columns)\n        ).mean(dim=1)\n\n        loss = criterion(\n            study_output,\n            targets\n        )\n\n        loss.backward()\n        optimizer.step()\n\n        train_loss += loss.item()\n\n    train_loss /= len(train_loader)\n\n\n    # =========================\n    # VALIDATION\n    # =========================\n    model.eval()\n\n    val_loss = 0.0\n\n    with torch.no_grad():\n\n        for images, targets in val_loader:\n\n            images = images.to(device, non_blocking=True)\n            targets = targets.to(device, non_blocking=True)\n\n            images_25d = make_25d(images)\n\n            B, N, C, H, W = images_25d.shape\n\n            cnn_input = images_25d.reshape(\n                B * N,\n                C,\n                H,\n                W\n            )\n\n            output = model(cnn_input)\n\n            study_output = output.reshape(\n                B,\n                N,\n                len(target_columns)\n            ).mean(dim=1)\n\n            loss = criterion(\n                study_output,\n                targets\n            )\n\n            val_loss += loss.item()\n\n    val_loss /= len(val_loader)\n\n\n    print(\n        f\"Epoch {epoch + 1:02d}/{num_epochs} | \"\n        f\"Train Loss: {train_loss:.4f} | \"\n        f\"Val Loss: {val_loss:.4f}\"\n    )\n\n\n    # =========================\n    # SAVE BEST MODEL\n    # =========================\n    if val_loss < best_val_loss:\n\n        best_val_loss = val_loss\n\n        torch.save(\n            model.state_dict(),\n            \"/kaggle/working/best_resnet18_knee.pth\"\n        )\n\n        print(\"  ✓ Best model saved\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:47:29.533343Z","iopub.execute_input":"2026-08-10T14:47:29.534204Z","iopub.status.idle":"2026-08-10T14:48:33.136241Z","shell.execute_reply.started":"2026-08-10T14:47:29.534167Z","shell.execute_reply":"2026-08-10T14:48:33.135348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.load_state_dict(\n    torch.load(\n        \"/kaggle/working/best_resnet18_knee.pth\",\n        map_location=device\n    )\n)\n\nmodel.eval()\n\nprint(\"Best model loaded\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:52:46.10031Z","iopub.execute_input":"2026-08-10T14:52:46.100793Z","iopub.status.idle":"2026-08-10T14:52:46.185156Z","shell.execute_reply.started":"2026-08-10T14:52:46.100757Z","shell.execute_reply":"2026-08-10T14:52:46.184246Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_predictions = []\nval_targets = []\n\nwith torch.no_grad():\n\n    for images, targets in val_loader:\n\n        images = images.to(device, non_blocking=True)\n\n        images_25d = make_25d(images)\n\n        B, N, C, H, W = images_25d.shape\n\n        cnn_input = images_25d.reshape(\n            B * N,\n            C,\n            H,\n            W\n        )\n\n        output = model(cnn_input)\n\n        study_output = output.reshape(\n            B,\n            N,\n            len(target_columns)\n        ).mean(dim=1)\n\n        probabilities = torch.sigmoid(\n            study_output\n        )\n\n        val_predictions.append(\n            probabilities.cpu().numpy()\n        )\n\n        val_targets.append(\n            targets.numpy()\n        )\n\nval_predictions = np.concatenate(\n    val_predictions,\n    axis=0\n)\n\nval_targets = np.concatenate(\n    val_targets,\n    axis=0\n)\n\nprint(\"Predictions:\", val_predictions.shape)\nprint(\"Targets:\", val_targets.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:52:57.823847Z","iopub.execute_input":"2026-08-10T14:52:57.824276Z","iopub.status.idle":"2026-08-10T14:52:59.296358Z","shell.execute_reply.started":"2026-08-10T14:52:57.824242Z","shell.execute_reply":"2026-08-10T14:52:59.295553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\n\nfor i, column in enumerate(target_columns):\n\n    y_true = val_targets[:, i]\n    y_pred = val_predictions[:, i]\n\n    try:\n        auc = roc_auc_score(y_true, y_pred)\n        print(f\"{column:20s} AUC: {auc:.4f}\")\n    except ValueError:\n        print(f\"{column:20s} AUC: unavailable\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T14:53:18.028174Z","iopub.execute_input":"2026-08-10T14:53:18.028482Z","iopub.status.idle":"2026-08-10T14:53:18.05728Z","shell.execute_reply.started":"2026-08-10T14:53:18.02845Z","shell.execute_reply":"2026-08-10T14:53:18.056432Z"}},"outputs":[],"execution_count":null}]}