{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# RSNA Knee Abnormality Detection — Dataset Discovery + EDA\n\nGoal for this notebook:\n1. Confirm the folder / file structure matches expectations\n2. Understand label coverage (gold-labeled vs report-only studies)\n3. Look at label distributions + co-occurrence\n4. Explore report text (language, length)\n5. Explore series-level metadata (plane, fluid-sensitive, fat-sat)\n6. Spot-check raw DICOM headers + pixel data\n7. Visualize a handful of slices\n\n","metadata":{}},{"cell_type":"code","source":"import os\nimport glob\nimport random\nfrom pathlib import Path\nfrom collections import Counter\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\npd.set_option(\"display.max_columns\", 50)\npd.set_option(\"display.width\", 160)\n\n# On Kaggle this is usually: /kaggle/input/rsna-knee-abnormality-detection\nDATA_DIR = Path(\"/kaggle/input/competitions/rsna-knee-abnormality-detection\")\n\nTRAIN_CSV = DATA_DIR / \"train.csv\"\nTRAIN_SERIES_CSV = DATA_DIR / \"train_series.csv\"\nTEST_CSV = DATA_DIR / \"test.csv\"\nTEST_SERIES_CSV = DATA_DIR / \"test_series.csv\"\nSAMPLE_SUB_CSV = DATA_DIR / \"sample_submission.csv\"\n\nTRAIN_SERIES_DIR = DATA_DIR / \"train_series\"\nTEST_SERIES_DIR = DATA_DIR / \"test_series\"\n\nfor p in [TRAIN_CSV, TRAIN_SERIES_CSV, TEST_CSV, TEST_SERIES_CSV, SAMPLE_SUB_CSV]:\n    print(f\"{'OK ' if p.exists() else 'MISSING'} -> {p}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:27.052152Z","iopub.execute_input":"2026-08-11T05:53:27.052511Z","iopub.status.idle":"2026-08-11T05:53:27.424213Z","shell.execute_reply.started":"2026-08-11T05:53:27.052474Z","shell.execute_reply":"2026-08-11T05:53:27.423072Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1. Load CSVs and do basic sanity checks","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(TRAIN_CSV)\ntrain_series = pd.read_csv(TRAIN_SERIES_CSV)\ntest = pd.read_csv(TEST_CSV)\ntest_series = pd.read_csv(TEST_SERIES_CSV)\nsample_sub = pd.read_csv(SAMPLE_SUB_CSV)\n\nprint(\"train.csv:\", train.shape)\nprint(\"train_series.csv:\", train_series.shape)\nprint(\"test.csv:\", test.shape)\nprint(\"test_series.csv:\", test_series.shape)\nprint(\"sample_submission.csv:\", sample_sub.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:27.427619Z","iopub.execute_input":"2026-08-11T05:53:27.428468Z","iopub.status.idle":"2026-08-11T05:53:27.721093Z","shell.execute_reply.started":"2026-08-11T05:53:27.428426Z","shell.execute_reply":"2026-08-11T05:53:27.720019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LABEL_COLS = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\",\n    \"Medial OA\", \"Lateral OA\", \"PF OA\",\n    \"Effusion\", \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\",\n]\n\ntrain.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:27.722599Z","iopub.execute_input":"2026-08-11T05:53:27.723115Z","iopub.status.idle":"2026-08-11T05:53:27.763096Z","shell.execute_reply.started":"2026-08-11T05:53:27.723076Z","shell.execute_reply":"2026-08-11T05:53:27.761957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_series.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:27.764706Z","iopub.execute_input":"2026-08-11T05:53:27.765155Z","iopub.status.idle":"2026-08-11T05:53:27.777615Z","shell.execute_reply.started":"2026-08-11T05:53:27.765123Z","shell.execute_reply":"2026-08-11T05:53:27.776541Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1a. Do CSV rows match folders on disk?\n\nConfirms `train.csv` studies line up with `train_series/` folders, and that\n`train_series.csv` rows line up with series subfolders. Catches path/schema\nsurprises early.\n","metadata":{}},{"cell_type":"code","source":"train_study_folders = set(p.name for p in TRAIN_SERIES_DIR.iterdir() if p.is_dir())\ncsv_study_ids = set(train[\"StudyInstanceUID\"])\n\nprint(\"Studies in train.csv        :\", len(csv_study_ids))\nprint(\"Study folders in train_series/:\", len(train_study_folders))\nprint(\"In CSV but no folder        :\", len(csv_study_ids - train_study_folders))\nprint(\"Folder but not in CSV       :\", len(train_study_folders - csv_study_ids))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:27.779223Z","iopub.execute_input":"2026-08-11T05:53:27.77959Z","iopub.status.idle":"2026-08-11T05:53:31.939836Z","shell.execute_reply.started":"2026-08-11T05:53:27.779563Z","shell.execute_reply":"2026-08-11T05:53:31.938736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Series-level check: for a sample of studies, confirm listed series folders exist\nsample_studies = random.sample(sorted(train_study_folders), min(20, len(train_study_folders)))\n\nmissing_series = []\nfor study_id in sample_studies:\n    study_dir = TRAIN_SERIES_DIR / study_id\n    on_disk = set(p.name for p in study_dir.iterdir() if p.is_dir())\n    expected = set(train_series.loc[train_series[\"StudyInstanceUID\"] == study_id, \"SeriesInstanceUID\"])\n    missing = expected - on_disk\n    if missing:\n        missing_series.append((study_id, missing))\n\nprint(f\"Checked {len(sample_studies)} studies, {len(missing_series)} had missing series folders\")\nmissing_series[:5]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:31.941177Z","iopub.execute_input":"2026-08-11T05:53:31.941681Z","iopub.status.idle":"2026-08-11T05:53:32.096415Z","shell.execute_reply.started":"2026-08-11T05:53:31.941652Z","shell.execute_reply":"2026-08-11T05:53:32.095406Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Label coverage: gold-labeled vs report-only\n\nThis is the single most important number for the whole competition plan —\nit tells us how much of the training set needs silver labels derived\nfrom the report text.\n","metadata":{}},{"cell_type":"code","source":"# A study is \"gold-labeled\" if all 12 label columns are non-null.\n# Adjust this logic if partially-labeled rows turn out to exist.\nis_fully_labeled = train[LABEL_COLS].notna().all(axis=1)\nis_any_labeled = train[LABEL_COLS].notna().any(axis=1)\n\nn_total = len(train)\nn_fully_labeled = is_fully_labeled.sum()\nn_partially_labeled = (is_any_labeled & ~is_fully_labeled).sum()\nn_unlabeled = (~is_any_labeled).sum()\n\nprint(f\"Total studies          : {n_total}\")\nprint(f\"Fully labeled          : {n_fully_labeled} ({n_fully_labeled/n_total:.1%})\")\nprint(f\"Partially labeled      : {n_partially_labeled} ({n_partially_labeled/n_total:.1%})\")\nprint(f\"No labels (report only): {n_unlabeled} ({n_unlabeled/n_total:.1%})\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:32.098891Z","iopub.execute_input":"2026-08-11T05:53:32.099234Z","iopub.status.idle":"2026-08-11T05:53:32.112892Z","shell.execute_reply.started":"2026-08-11T05:53:32.099204Z","shell.execute_reply":"2026-08-11T05:53:32.111771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Does every row have a report, regardless of label status?\nhas_report = train[\"Report\"].notna() & (train[\"Report\"].str.strip() != \"\")\nprint(f\"Studies with non-empty Report text: {has_report.sum()} / {n_total} ({has_report.mean():.1%})\")\nprint(f\"Fully-labeled studies WITHOUT a report: {(is_fully_labeled & ~has_report).sum()}\")\nprint(f\"Unlabeled studies WITHOUT a report (dead weight, unusable): {(~is_any_labeled & ~has_report).sum()}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:32.114128Z","iopub.execute_input":"2026-08-11T05:53:32.114859Z","iopub.status.idle":"2026-08-11T05:53:32.138158Z","shell.execute_reply.started":"2026-08-11T05:53:32.114831Z","shell.execute_reply":"2026-08-11T05:53:32.136987Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Label distribution (gold-labeled subset only)","metadata":{}},{"cell_type":"code","source":"gold = train[is_fully_labeled].copy()\n\nlabel_prevalence = gold[LABEL_COLS].mean().sort_values(ascending=False)\nprint(\"Positive rate per label (gold-labeled studies):\")\nprint(label_prevalence)\n\nfig, ax = plt.subplots(figsize=(9, 5))\nlabel_prevalence.plot(kind=\"barh\", ax=ax)\nax.set_xlabel(\"Positive rate\")\nax.set_title(f\"Label prevalence (n={len(gold)} gold-labeled studies)\")\nax.invert_yaxis()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:40.879903Z","iopub.execute_input":"2026-08-11T05:53:40.880513Z","iopub.status.idle":"2026-08-11T05:53:41.296509Z","shell.execute_reply.started":"2026-08-11T05:53:40.880478Z","shell.execute_reply":"2026-08-11T05:53:41.2954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# How many positive labels does a study typically have? (multi-label density)\nn_positive_per_study = gold[LABEL_COLS].sum(axis=1)\nprint(n_positive_per_study.describe())\n\nfig, ax = plt.subplots(figsize=(7, 4))\nn_positive_per_study.value_counts().sort_index().plot(kind=\"bar\", ax=ax)\nax.set_xlabel(\"Number of positive findings in a study\")\nax.set_ylabel(\"Count of studies\")\nax.set_title(\"How many findings co-occur per study\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:41.29836Z","iopub.execute_input":"2026-08-11T05:53:41.298735Z","iopub.status.idle":"2026-08-11T05:53:41.661339Z","shell.execute_reply.started":"2026-08-11T05:53:41.298704Z","shell.execute_reply":"2026-08-11T05:53:41.659915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Co-occurrence matrix between labels (correlation)\nimport seaborn as sns  # if unavailable on Kaggle image, pip install seaborn\n\ncorr = gold[LABEL_COLS].corr()\nfig, ax = plt.subplots(figsize=(9, 7))\nsns.heatmap(corr, annot=True, fmt=\".2f\", cmap=\"coolwarm\", center=0, ax=ax)\nax.set_title(\"Label co-occurrence (correlation)\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:41.663173Z","iopub.execute_input":"2026-08-11T05:53:41.66391Z","iopub.status.idle":"2026-08-11T05:53:43.525608Z","shell.execute_reply.started":"2026-08-11T05:53:41.663876Z","shell.execute_reply":"2026-08-11T05:53:43.524181Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Report text exploration\n\nLanguage mix and report length matter a lot for Week 2-3 label-extraction\nwork (regex/keyword pass) and the later LLM-based extraction pass.\n","metadata":{}},{"cell_type":"code","source":"train[\"report_len_chars\"] = train[\"Report\"].fillna(\"\").str.len()\ntrain[\"report_len_words\"] = train[\"Report\"].fillna(\"\").str.split().str.len()\n\nprint(train[\"report_len_words\"].describe())\n\nfig, ax = plt.subplots(figsize=(7, 4))\ntrain.loc[has_report, \"report_len_words\"].hist(bins=50, ax=ax)\nax.set_xlabel(\"Report length (words)\")\nax.set_title(\"Report length distribution\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:43.527732Z","iopub.execute_input":"2026-08-11T05:53:43.528211Z","iopub.status.idle":"2026-08-11T05:53:43.919505Z","shell.execute_reply.started":"2026-08-11T05:53:43.528182Z","shell.execute_reply":"2026-08-11T05:53:43.918352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Rough language detection on a sample (install if needed: pip install langdetect)\nfrom langdetect import detect, DetectorFactory\nDetectorFactory.seed = 0\n\nsample_reports = train.loc[has_report, \"Report\"].sample(min(300, has_report.sum()), random_state=42)\n\ndef safe_detect(text):\n    try:\n        return detect(text)\n    except Exception:\n        return \"unknown\"\n\nlangs = sample_reports.apply(safe_detect)\nlang_counts = langs.value_counts()\nprint(lang_counts)\n\nfig, ax = plt.subplots(figsize=(7, 4))\nlang_counts.plot(kind=\"bar\", ax=ax)\nax.set_title(f\"Detected report language (sample of {len(sample_reports)})\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:43.920757Z","iopub.execute_input":"2026-08-11T05:53:43.92115Z","iopub.status.idle":"2026-08-11T05:53:46.6052Z","shell.execute_reply.started":"2026-08-11T05:53:43.921105Z","shell.execute_reply":"2026-08-11T05:53:46.604226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Does report length or presence differ between gold-labeled and unlabeled studies?\n# (checks for sampling bias in which studies got gold labels)\ncomparison = pd.DataFrame({\n    \"gold_labeled\": train.loc[is_fully_labeled, \"report_len_words\"].describe(),\n    \"unlabeled\": train.loc[~is_any_labeled, \"report_len_words\"].describe(),\n})\ncomparison\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:46.607158Z","iopub.execute_input":"2026-08-11T05:53:46.607573Z","iopub.status.idle":"2026-08-11T05:53:46.627658Z","shell.execute_reply.started":"2026-08-11T05:53:46.60753Z","shell.execute_reply":"2026-08-11T05:53:46.626151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Eyeball a few raw reports\nfor i, row in train.loc[has_report].sample(3, random_state=1).iterrows():\n    print(\"=\" * 80)\n    print(\"StudyInstanceUID:\", row[\"StudyInstanceUID\"])\n    print(\"Labeled:\", bool(is_fully_labeled.loc[i]))\n    print(row[\"Report\"][:800])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:46.628966Z","iopub.execute_input":"2026-08-11T05:53:46.62926Z","iopub.status.idle":"2026-08-11T05:53:46.680616Z","shell.execute_reply.started":"2026-08-11T05:53:46.629233Z","shell.execute_reply":"2026-08-11T05:53:46.679085Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Series-level metadata (`train_series.csv`)","metadata":{}},{"cell_type":"code","source":"print(\"Anatomical_Plane distribution:\")\nprint(train_series[\"Anatomical_Plane\"].value_counts(dropna=False))\n\nprint(\"\\nFluid_Sensitive distribution:\")\nprint(train_series[\"Fluid_Sensitive\"].value_counts(dropna=False))\n\nprint(\"\\nFat_Suppression distribution:\")\nprint(train_series[\"Fat_Suppression\"].value_counts(dropna=False))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:46.682235Z","iopub.execute_input":"2026-08-11T05:53:46.682721Z","iopub.status.idle":"2026-08-11T05:53:46.703681Z","shell.execute_reply.started":"2026-08-11T05:53:46.682673Z","shell.execute_reply":"2026-08-11T05:53:46.702214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cross-tab of plane x fluid-sensitive x fat-sat -> the \"vocabulary\" of sequence types\nseq_type_counts = (\n    train_series\n    .groupby([\"Anatomical_Plane\", \"Fluid_Sensitive\", \"Fat_Suppression\"])\n    .size()\n    .reset_index(name=\"count\")\n    .sort_values(\"count\", ascending=False)\n)\nseq_type_counts\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:46.763248Z","iopub.execute_input":"2026-08-11T05:53:46.76359Z","iopub.status.idle":"2026-08-11T05:53:46.786755Z","shell.execute_reply.started":"2026-08-11T05:53:46.763564Z","shell.execute_reply":"2026-08-11T05:53:46.785478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Series per study distribution\nseries_per_study = train_series.groupby(\"StudyInstanceUID\").size()\nprint(series_per_study.describe())\n\nfig, ax = plt.subplots(figsize=(7, 4))\nseries_per_study.hist(bins=30, ax=ax)\nax.set_xlabel(\"Number of series per study\")\nax.set_title(\"Series-per-study distribution\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:46.950698Z","iopub.execute_input":"2026-08-11T05:53:46.951067Z","iopub.status.idle":"2026-08-11T05:53:47.15299Z","shell.execute_reply.started":"2026-08-11T05:53:46.951041Z","shell.execute_reply":"2026-08-11T05:53:47.15159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Do all studies have at least one sagittal series? Coronal? This matters for\n# ACL (sagittal) / meniscus (sagittal+coronal) modeling decisions later.\nplane_presence = (\n    train_series\n    .assign(has_plane=1)\n    .pivot_table(index=\"StudyInstanceUID\", columns=\"Anatomical_Plane\", values=\"has_plane\",\n                 aggfunc=\"max\", fill_value=0)\n)\nprint(\"Fraction of studies with each plane present at least once:\")\nprint(plane_presence.mean())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:47.220743Z","iopub.execute_input":"2026-08-11T05:53:47.221054Z","iopub.status.idle":"2026-08-11T05:53:47.269135Z","shell.execute_reply.started":"2026-08-11T05:53:47.221028Z","shell.execute_reply":"2026-08-11T05:53:47.268025Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. DICOM header + pixel spot-check\n\nPulling a small random sample of raw `.dcm` files to look at transfer syntax,\npixel array shape, intensity range, and orientation. This is NOT meant to\nprocess the full dataset — just enough to inform Week 2 preprocessing design.\n","metadata":{}},{"cell_type":"code","source":"import pydicom\n\ndef sample_dcm_paths(n=15, seed=0):\n    rng = random.Random(seed)\n    study_ids = rng.sample(sorted(train_study_folders), min(n, len(train_study_folders)))\n    paths = []\n    for sid in study_ids:\n        study_dir = TRAIN_SERIES_DIR / sid\n        series_dirs = [p for p in study_dir.iterdir() if p.is_dir()]\n        if not series_dirs:\n            continue\n        series_dir = rng.choice(series_dirs)\n        dcm_files = list(series_dir.glob(\"*.dcm\"))\n        if not dcm_files:\n            continue\n        paths.append(rng.choice(dcm_files))\n    return paths\n\ndcm_paths = sample_dcm_paths(15)\nprint(f\"Sampled {len(dcm_paths)} DICOM files\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:47.696689Z","iopub.execute_input":"2026-08-11T05:53:47.697822Z","iopub.status.idle":"2026-08-11T05:53:48.667536Z","shell.execute_reply.started":"2026-08-11T05:53:47.697788Z","shell.execute_reply":"2026-08-11T05:53:48.666017Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rows = []\nfor p in dcm_paths:\n    ds = pydicom.dcmread(p)\n    arr = ds.pixel_array\n    rows.append({\n        \"path\": str(p.relative_to(TRAIN_SERIES_DIR)),\n        \"TransferSyntaxUID\": str(ds.file_meta.TransferSyntaxUID) if hasattr(ds, \"file_meta\") else None,\n        \"shape\": arr.shape,\n        \"dtype\": str(arr.dtype),\n        \"min\": arr.min(),\n        \"max\": arr.max(),\n        \"PixelSpacing\": getattr(ds, \"PixelSpacing\", None),\n        \"SliceThickness\": getattr(ds, \"SliceThickness\", None),\n        \"Modality\": getattr(ds, \"Modality\", None),\n    })\n\ndcm_info = pd.DataFrame(rows)\ndcm_info\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:48.669143Z","iopub.execute_input":"2026-08-11T05:53:48.669684Z","iopub.status.idle":"2026-08-11T05:53:48.884206Z","shell.execute_reply.started":"2026-08-11T05:53:48.669653Z","shell.execute_reply":"2026-08-11T05:53:48.883226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Unique transfer syntaxes seen in this sample:\")\nprint(dcm_info[\"TransferSyntaxUID\"].value_counts())\n\nprint(\"\\nIntensity range summary:\")\nprint(dcm_info[[\"min\", \"max\"]].describe())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:50.61475Z","iopub.execute_input":"2026-08-11T05:53:50.615615Z","iopub.status.idle":"2026-08-11T05:53:50.628152Z","shell.execute_reply.started":"2026-08-11T05:53:50.615583Z","shell.execute_reply":"2026-08-11T05:53:50.626727Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. Visual sanity check: render a few slices across plane/sequence types","metadata":{}},{"cell_type":"code","source":"def load_middle_slice(series_dir):\n    dcm_files = sorted(series_dir.glob(\"*.dcm\"))\n    if not dcm_files:\n        return None\n    mid = dcm_files[len(dcm_files) // 2]\n    ds = pydicom.dcmread(mid)\n    return ds.pixel_array\n\n# Grab one series per Anatomical_Plane to eyeball\nplanes_seen = {}\nfor _, row in train_series.sample(frac=1, random_state=3).iterrows():\n    plane = row[\"Anatomical_Plane\"]\n    if plane in planes_seen:\n        continue\n    series_dir = TRAIN_SERIES_DIR / row[\"StudyInstanceUID\"] / row[\"SeriesInstanceUID\"]\n    if series_dir.exists():\n        planes_seen[plane] = series_dir\n    if len(planes_seen) >= 3:\n        break\n\nfig, axes = plt.subplots(1, len(planes_seen), figsize=(5 * len(planes_seen), 5))\nif len(planes_seen) == 1:\n    axes = [axes]\nfor ax, (plane, series_dir) in zip(axes, planes_seen.items()):\n    img = load_middle_slice(series_dir)\n    if img is not None:\n        ax.imshow(img, cmap=\"gray\")\n    ax.set_title(plane)\n    ax.axis(\"off\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T05:53:52.435154Z","iopub.execute_input":"2026-08-11T05:53:52.435547Z","iopub.status.idle":"2026-08-11T05:53:53.152323Z","shell.execute_reply.started":"2026-08-11T05:53:52.435518Z","shell.execute_reply":"2026-08-11T05:53:53.150719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}