{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"57fc5a24-d33b-4526-b14b-f02e14d95863","cell_type":"markdown","source":"# Exploratory Data Analysis for RSNA Knee Abnormality Detection Dataset \n\nCovers:\n1. Table loading and schema check\n2. Label coverage and missingness (only a subset of studies is labeled)\n3. Label prevalence and co-occurrence\n4. Radiology report text (multilingual, training-time only)\n5. Series-level metadata (plane, fluid sensitivity, fat suppression)\n6. Train/Test composition\n7. DICOM header sampling (transfer syntax, spacing, matrix size)\n8. Pixel data visualization\n9. Test set sanity checks\n","metadata":{}},{"id":"cf666838-8c8f-410f-8f65-38b303a5a10f","cell_type":"code","source":"import os\nimport glob\nimport warnings\nfrom collections import Counter\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nwarnings.filterwarnings(\"ignore\")\npd.set_option(\"display.max_columns\", 50)\npd.set_option(\"display.width\", 160)\nsns.set_theme(style=\"whitegrid\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:53.706685Z","iopub.execute_input":"2026-08-20T10:10:53.707593Z","iopub.status.idle":"2026-08-20T10:10:53.718866Z","shell.execute_reply.started":"2026-08-20T10:10:53.707557Z","shell.execute_reply":"2026-08-20T10:10:53.717735Z"}},"outputs":[],"execution_count":null},{"id":"0b24520a-60ef-4b0e-9fd5-6dd6dce22858","cell_type":"code","source":"CANDIDATE_ROOTS = [\n    \"/kaggle/input/competitions/rsna-knee-abnormality-detection\",\n    \"/kaggle/input/rsna-knee-abnormality-detection\",\n]\n\nDATA_ROOT = None\nfor root in CANDIDATE_ROOTS:\n    if os.path.exists(os.path.join(root, \"train.csv\")):\n        DATA_ROOT = root\n        break\n\nif DATA_ROOT is None:\n    raise FileNotFoundError(f\"train.csv not found under any of: {CANDIDATE_ROOTS}\")\n\nprint(\"DATA_ROOT:\", DATA_ROOT)\nprint(sorted(os.listdir(DATA_ROOT)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:53.720436Z","iopub.execute_input":"2026-08-20T10:10:53.720865Z","iopub.status.idle":"2026-08-20T10:10:53.751691Z","shell.execute_reply.started":"2026-08-20T10:10:53.720838Z","shell.execute_reply":"2026-08-20T10:10:53.750312Z"}},"outputs":[],"execution_count":null},{"id":"99a92c3f-f499-4f86-af3e-98db1bedf800","cell_type":"markdown","source":"## 1. Load tables","metadata":{}},{"id":"6584750f-7e17-4b7c-88e1-fadd459b9637","cell_type":"code","source":"train = pd.read_csv(os.path.join(DATA_ROOT, \"train.csv\"))\ntrain_series = pd.read_csv(os.path.join(DATA_ROOT, \"train_series.csv\"))\ntest = pd.read_csv(os.path.join(DATA_ROOT, \"test.csv\"))\ntest_series = pd.read_csv(os.path.join(DATA_ROOT, \"test_series.csv\"))\nsample_submission = pd.read_csv(os.path.join(DATA_ROOT, \"sample_submission.csv\"))\n\nLABEL_COLS = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\",\n    \"Medial OA\", \"Lateral OA\", \"PF OA\", \"Effusion\",\n    \"Synovitis\", \"Baker's\", \"Contusion\", \"Fracture\",\n]\n\nprint(\"train:\", train.shape)\nprint(\"train_series:\", train_series.shape)\nprint(\"test:\", test.shape)\nprint(\"test_series:\", test_series.shape)\nprint(\"sample_submission:\", sample_submission.shape)\nassert list(sample_submission.columns) == [\"StudyInstanceUID\"] + LABEL_COLS\ntrain.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:53.752959Z","iopub.execute_input":"2026-08-20T10:10:53.753353Z","iopub.status.idle":"2026-08-20T10:10:53.949315Z","shell.execute_reply.started":"2026-08-20T10:10:53.753308Z","shell.execute_reply":"2026-08-20T10:10:53.948261Z"}},"outputs":[],"execution_count":null},{"id":"d825524d-8124-45e5-9f22-13d69a7053dc","cell_type":"markdown","source":"## 2. Label coverage and missingness\n\nOnly a subset of studies carries per-condition labels. Confirm whether labels are all-or-nothing per study or partially filled.","metadata":{}},{"id":"b81f6fe8-78eb-4ee0-813d-116204bba1ef","cell_type":"code","source":"n_total = len(train)\nall_null = train[LABEL_COLS].isna().all(axis=1)\nall_present = train[LABEL_COLS].notna().all(axis=1)\npartial = ~(all_null | all_present)\n\nprint(f\"total studies:        {n_total}\")\nprint(f\"fully unlabeled:      {all_null.sum()} ({all_null.mean():.1%})\")\nprint(f\"fully labeled:        {all_present.sum()} ({all_present.mean():.1%})\")\nprint(f\"partially labeled:    {partial.sum()} ({partial.mean():.1%})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:53.951597Z","iopub.execute_input":"2026-08-20T10:10:53.95257Z","iopub.status.idle":"2026-08-20T10:10:53.966464Z","shell.execute_reply.started":"2026-08-20T10:10:53.952538Z","shell.execute_reply":"2026-08-20T10:10:53.96551Z"}},"outputs":[],"execution_count":null},{"id":"83267eeb-2321-4c32-bd35-1b991677dec5","cell_type":"code","source":"missing_per_col = train[LABEL_COLS].isna().sum().sort_values(ascending=False)\n\nfig, ax = plt.subplots(figsize=(9, 4))\n(n_total - missing_per_col).plot(kind=\"bar\", ax=ax, color=\"steelblue\")\nax.set_ylabel(\"studies with non-null label\")\nax.set_title(\"Label coverage per column\")\nplt.xticks(rotation=45, ha=\"right\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:53.967789Z","iopub.execute_input":"2026-08-20T10:10:53.968143Z","iopub.status.idle":"2026-08-20T10:10:54.344013Z","shell.execute_reply.started":"2026-08-20T10:10:53.968109Z","shell.execute_reply":"2026-08-20T10:10:54.342924Z"}},"outputs":[],"execution_count":null},{"id":"38243f3a-5ccc-4297-a748-89583ce299c2","cell_type":"markdown","source":"## 3. Label prevalence and co-occurrence (labeled subset only)","metadata":{}},{"id":"f137c520-e131-46a1-996a-e5bd3c0be19e","cell_type":"code","source":"labeled = train.loc[all_present].copy()\nlabeled[LABEL_COLS] = labeled[LABEL_COLS].astype(int)\nprint(\"labeled studies:\", len(labeled))\n\nprevalence = labeled[LABEL_COLS].mean().sort_values(ascending=False)\nfig, ax = plt.subplots(figsize=(9, 4))\nprevalence.plot(kind=\"bar\", ax=ax, color=\"indianred\")\nax.set_ylabel(\"positive rate\")\nax.set_title(\"Label prevalence in labeled subset\")\nplt.xticks(rotation=45, ha=\"right\")\nplt.tight_layout()\nplt.show()\nprevalence","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:54.345279Z","iopub.execute_input":"2026-08-20T10:10:54.345669Z","iopub.status.idle":"2026-08-20T10:10:54.557613Z","shell.execute_reply.started":"2026-08-20T10:10:54.345645Z","shell.execute_reply":"2026-08-20T10:10:54.556652Z"}},"outputs":[],"execution_count":null},{"id":"c7d7d584-0320-4fcd-a0a9-5f309c95cf7a","cell_type":"code","source":"corr = labeled[LABEL_COLS].corr()\nfig, ax = plt.subplots(figsize=(8, 7))\nsns.heatmap(corr, annot=True, fmt=\".2f\", cmap=\"coolwarm\", center=0, ax=ax)\nax.set_title(\"Label correlation (labeled subset)\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:54.558865Z","iopub.execute_input":"2026-08-20T10:10:54.559208Z","iopub.status.idle":"2026-08-20T10:10:55.126905Z","shell.execute_reply.started":"2026-08-20T10:10:54.559184Z","shell.execute_reply":"2026-08-20T10:10:55.125656Z"}},"outputs":[],"execution_count":null},{"id":"95f1d772-9750-4a98-bf81-6c3f2bc93d14","cell_type":"code","source":"n_positive_labels = labeled[LABEL_COLS].sum(axis=1)\nfig, ax = plt.subplots(figsize=(7, 4))\nn_positive_labels.value_counts().sort_index().plot(kind=\"bar\", ax=ax, color=\"seagreen\")\nax.set_xlabel(\"number of positive labels in a study\")\nax.set_ylabel(\"study count\")\nax.set_title(\"Multi-label burden per study\")\nplt.tight_layout()\nplt.show()\nprint(n_positive_labels.describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:55.128107Z","iopub.execute_input":"2026-08-20T10:10:55.128632Z","iopub.status.idle":"2026-08-20T10:10:55.30962Z","shell.execute_reply.started":"2026-08-20T10:10:55.128585Z","shell.execute_reply":"2026-08-20T10:10:55.308631Z"}},"outputs":[],"execution_count":null},{"id":"0037f36f-dcab-4084-b167-02e9c5b3683e","cell_type":"markdown","source":"## 4. Radiology report text\n\nReports are provided for training only; the competition states the `Report` field will not be present at test time, so any text signal has to be used during training (weak labeling of the unlabeled studies, auxiliary losses, distillation) rather than as a live inference input.","metadata":{}},{"id":"c89c54ae-7217-4f90-be4c-b4cdc754bb76","cell_type":"code","source":"report_present = train[\"Report\"].notna() & (train[\"Report\"].str.strip() != \"\")\nprint(f\"studies with a report: {report_present.sum()} ({report_present.mean():.1%})\")\nprint(f\"labeled AND has report:     {(report_present & all_present).sum()}\")\nprint(f\"labeled AND report missing: {(~report_present & all_present).sum()}\")\nprint(f\"unlabeled AND has report:   {(report_present & all_null).sum()}\")\nprint(f\"unlabeled AND report missing: {(~report_present & all_null).sum()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:55.310491Z","iopub.execute_input":"2026-08-20T10:10:55.310745Z","iopub.status.idle":"2026-08-20T10:10:55.327844Z","shell.execute_reply.started":"2026-08-20T10:10:55.310722Z","shell.execute_reply":"2026-08-20T10:10:55.32673Z"}},"outputs":[],"execution_count":null},{"id":"e4754ed6-3b17-4d5f-bab9-1452e4af2e1f","cell_type":"code","source":"report_words = train.loc[report_present, \"Report\"].str.split().str.len()\n\nfig, ax = plt.subplots(figsize=(8, 4))\nreport_words.clip(upper=report_words.quantile(0.99)).hist(bins=50, ax=ax, color=\"slateblue\")\nax.set_xlabel(\"word count\")\nax.set_title(\"Report length distribution (99th pct clipped)\")\nplt.tight_layout()\nplt.show()\nreport_words.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:55.330993Z","iopub.execute_input":"2026-08-20T10:10:55.331423Z","iopub.status.idle":"2026-08-20T10:10:55.732812Z","shell.execute_reply.started":"2026-08-20T10:10:55.331392Z","shell.execute_reply":"2026-08-20T10:10:55.73188Z"}},"outputs":[],"execution_count":null},{"id":"f74630ac-b59a-4c8a-a1d7-33c4b5b38c6d","cell_type":"code","source":"LANG_STOPWORDS = {\n    \"en\": {\"the\", \"and\", \"with\", \"no\", \"of\", \"is\", \"normal\", \"there\", \"left\", \"right\"},\n    \"es\": {\"de\", \"la\", \"el\", \"con\", \"sin\", \"los\", \"las\", \"impresion\", \"hallazgos\", \"rodilla\"},\n    \"pt\": {\"de\", \"da\", \"do\", \"com\", \"sem\", \"os\", \"as\", \"joelho\", \"achados\"},\n    \"nl\": {\"van\", \"met\", \"geen\", \"rechts\", \"links\", \"bevindingen\", \"knie\"},\n    \"de\": {\"der\", \"die\", \"das\", \"mit\", \"ohne\", \"rechts\", \"links\", \"knie\", \"befund\"},\n    \"fr\": {\"de\", \"le\", \"la\", \"avec\", \"sans\", \"genou\", \"conclusion\"},\n    \"it\": {\"di\", \"con\", \"senza\", \"ginocchio\", \"referto\"},\n}\n\ndef guess_language(text):\n    tokens = set(w.strip(\".,;:()[]\").lower() for w in str(text).split())\n    scores = {lang: len(tokens & words) for lang, words in LANG_STOPWORDS.items()}\n    best = max(scores, key=scores.get)\n    return best if scores[best] > 0 else \"unknown\"\n\nsample_reports = train.loc[report_present, \"Report\"].sample(min(3000, report_present.sum()), random_state=0)\nlang_counts = Counter(guess_language(t) for t in sample_reports)\n\nfig, ax = plt.subplots(figsize=(7, 4))\npd.Series(lang_counts).sort_values(ascending=False).plot(kind=\"bar\", ax=ax, color=\"darkorange\")\nax.set_title(\"Heuristic language distribution (stopword-based, approximate)\")\nax.set_ylabel(\"sampled reports\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:55.734202Z","iopub.execute_input":"2026-08-20T10:10:55.734633Z","iopub.status.idle":"2026-08-20T10:10:56.090462Z","shell.execute_reply.started":"2026-08-20T10:10:55.734595Z","shell.execute_reply":"2026-08-20T10:10:56.089504Z"}},"outputs":[],"execution_count":null},{"id":"97685942-93b0-4968-b38a-5ca5f44ac238","cell_type":"markdown","source":"## 5. Series-level metadata","metadata":{}},{"id":"1d1ab87f-e901-4208-b0d8-febc77330916","cell_type":"code","source":"series_per_study = train_series.groupby(\"StudyInstanceUID\").size()\nfig, ax = plt.subplots(figsize=(7, 4))\nseries_per_study.plot(kind=\"hist\", bins=30, ax=ax, color=\"teal\")\nax.set_xlabel(\"series per study\")\nax.set_title(\"Series count per study (train)\")\nplt.tight_layout()\nplt.show()\nseries_per_study.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:56.091628Z","iopub.execute_input":"2026-08-20T10:10:56.09195Z","iopub.status.idle":"2026-08-20T10:10:56.294192Z","shell.execute_reply.started":"2026-08-20T10:10:56.091912Z","shell.execute_reply":"2026-08-20T10:10:56.293186Z"}},"outputs":[],"execution_count":null},{"id":"c53d05f8-b196-447a-8633-339690f52a0b","cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(15, 4))\ntrain_series[\"Anatomical_Plane\"].value_counts().plot(kind=\"bar\", ax=axes[0], color=\"cornflowerblue\")\naxes[0].set_title(\"Anatomical_Plane\")\ntrain_series[\"Fluid_Sensitive\"].value_counts().plot(kind=\"bar\", ax=axes[1], color=\"salmon\")\naxes[1].set_title(\"Fluid_Sensitive\")\ntrain_series[\"Fat_Suppression\"].value_counts().plot(kind=\"bar\", ax=axes[2], color=\"mediumseagreen\")\naxes[2].set_title(\"Fat_Suppression\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:56.295478Z","iopub.execute_input":"2026-08-20T10:10:56.295876Z","iopub.status.idle":"2026-08-20T10:10:56.679149Z","shell.execute_reply.started":"2026-08-20T10:10:56.295849Z","shell.execute_reply":"2026-08-20T10:10:56.678171Z"}},"outputs":[],"execution_count":null},{"id":"89c81c23-ce75-4135-966b-3ef366b3df4d","cell_type":"code","source":"crosstab = pd.crosstab(train_series[\"Fluid_Sensitive\"], train_series[\"Fat_Suppression\"])\nprint(crosstab)\n\nprotocol = (\n    train_series\n    .groupby([\"Anatomical_Plane\", \"Fluid_Sensitive\", \"Fat_Suppression\"])\n    .size()\n    .sort_values(ascending=False)\n)\nprotocol.head(20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:56.680393Z","iopub.execute_input":"2026-08-20T10:10:56.68072Z","iopub.status.idle":"2026-08-20T10:10:56.707211Z","shell.execute_reply.started":"2026-08-20T10:10:56.680696Z","shell.execute_reply":"2026-08-20T10:10:56.706196Z"}},"outputs":[],"execution_count":null},{"id":"b8ebd82d-ec7f-4e94-98f6-f644c8bb45b8","cell_type":"markdown","source":"## 6. Train vs test series composition\n\nCheck whether the protocol mix (plane, fluid sensitivity, fat suppression, series count) is similar between labeled train, unlabeled train, and test, as a rough domain-shift check.","metadata":{}},{"id":"72ed154f-cbdd-4fed-af05-06a267d5c9e9","cell_type":"code","source":"labeled_ids = set(train.loc[all_present, \"StudyInstanceUID\"])\nunlabeled_ids = set(train.loc[all_null, \"StudyInstanceUID\"])\n\ndef plane_distribution(series_df, ids=None):\n    df = series_df if ids is None else series_df[series_df[\"StudyInstanceUID\"].isin(ids)]\n    return df[\"Anatomical_Plane\"].value_counts(normalize=True)\n\ncomparison = pd.DataFrame({\n    \"labeled_train\": plane_distribution(train_series, labeled_ids),\n    \"unlabeled_train\": plane_distribution(train_series, unlabeled_ids),\n    \"test\": plane_distribution(test_series),\n})\ncomparison","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:56.708655Z","iopub.execute_input":"2026-08-20T10:10:56.709223Z","iopub.status.idle":"2026-08-20T10:10:56.74031Z","shell.execute_reply.started":"2026-08-20T10:10:56.709182Z","shell.execute_reply":"2026-08-20T10:10:56.739422Z"}},"outputs":[],"execution_count":null},{"id":"8fb89a67-1337-494d-a46d-53a5a4c663d5","cell_type":"markdown","source":"## 7. DICOM header sampling\n\nLoad headers only (no pixel data) for a sample of series to check transfer syntax, matrix size, and spacing before committing to a preprocessing pipeline.","metadata":{}},{"id":"4c96c3b8-7cf1-4ad8-911b-88fddb496259","cell_type":"code","source":"import pydicom\n\nTRAIN_SERIES_DIR = os.path.join(DATA_ROOT, \"train_series\")\n\ndef list_series_dirs(study_id):\n    study_dir = os.path.join(TRAIN_SERIES_DIR, study_id)\n    return [os.path.join(study_dir, s) for s in os.listdir(study_dir)]\n\ndef first_dcm(series_dir):\n    files = sorted(glob.glob(os.path.join(series_dir, \"*.dcm\")))\n    return files[0] if files else None\n\nsample_studies = train_series[\"StudyInstanceUID\"].drop_duplicates().sample(30, random_state=0).tolist()\n\nheader_rows = []\nfor study_id in sample_studies:\n    for series_dir in list_series_dirs(study_id):\n        f = first_dcm(series_dir)\n        if f is None:\n            continue\n        ds = pydicom.dcmread(f, stop_before_pixels=True)\n        n_slices = len(glob.glob(os.path.join(series_dir, \"*.dcm\")))\n        header_rows.append({\n            \"StudyInstanceUID\": study_id,\n            \"SeriesInstanceUID\": os.path.basename(series_dir),\n            \"n_slices\": n_slices,\n            \"Rows\": getattr(ds, \"Rows\", None),\n            \"Columns\": getattr(ds, \"Columns\", None),\n            \"PixelSpacing\": tuple(ds.PixelSpacing) if \"PixelSpacing\" in ds else None,\n            \"SliceThickness\": getattr(ds, \"SliceThickness\", None),\n            \"SpacingBetweenSlices\": getattr(ds, \"SpacingBetweenSlices\", None),\n            \"BitsAllocated\": getattr(ds, \"BitsAllocated\", None),\n            \"TransferSyntaxUID\": str(ds.file_meta.TransferSyntaxUID) if \"TransferSyntaxUID\" in ds.file_meta else None,\n            \"Manufacturer\": getattr(ds, \"Manufacturer\", None),\n        })\n\nheaders_df = pd.DataFrame(header_rows)\nprint(headers_df.shape)\nheaders_df.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:56.741689Z","iopub.execute_input":"2026-08-20T10:10:56.742051Z","iopub.status.idle":"2026-08-20T10:10:57.575872Z","shell.execute_reply.started":"2026-08-20T10:10:56.742015Z","shell.execute_reply":"2026-08-20T10:10:57.574765Z"}},"outputs":[],"execution_count":null},{"id":"1a52273c-acd5-4043-9ce7-2cf53b64e67a","cell_type":"code","source":"print(\"Transfer syntax distribution:\")\nprint(headers_df[\"TransferSyntaxUID\"].value_counts())\nprint()\nprint(\"Matrix size distribution:\")\nprint(headers_df[[\"Rows\", \"Columns\"]].value_counts().head(10))\nprint()\nprint(\"Slices per series:\")\nprint(headers_df[\"n_slices\"].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:57.577183Z","iopub.execute_input":"2026-08-20T10:10:57.578219Z","iopub.status.idle":"2026-08-20T10:10:57.592024Z","shell.execute_reply.started":"2026-08-20T10:10:57.578185Z","shell.execute_reply":"2026-08-20T10:10:57.590901Z"}},"outputs":[],"execution_count":null},{"id":"8e1bb786-b482-46bc-a1ea-40cfa3cd06aa","cell_type":"markdown","source":"## 8. Pixel data visualization\n\nJPEG Lossless / JPEG 2000 transfer syntaxes need `pylibjpeg` or `gdcm` plugins to decode. Try installing them; if there is no internet access on this kernel, decoding falls back gracefully and only uncompressed series will render.","metadata":{}},{"id":"6cefd46b-3b6b-46a5-9684-429094f0035f","cell_type":"code","source":"try:\n    import pylibjpeg  # noqa: F401\n    HAS_DECODER = True\nexcept ImportError:\n    HAS_DECODER = False\n\nif not HAS_DECODER:\n    os.system(\"pip install -q pylibjpeg pylibjpeg-libjpeg pylibjpeg-openjpeg python-gdcm\")\n    try:\n        import pylibjpeg  # noqa: F401\n        HAS_DECODER = True\n    except ImportError:\n        HAS_DECODER = False\n\nprint(\"decoder available:\", HAS_DECODER)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:57.59335Z","iopub.execute_input":"2026-08-20T10:10:57.59371Z","iopub.status.idle":"2026-08-20T10:10:57.612427Z","shell.execute_reply.started":"2026-08-20T10:10:57.593673Z","shell.execute_reply":"2026-08-20T10:10:57.611529Z"}},"outputs":[],"execution_count":null},{"id":"63ed0ae4-2be7-407f-b005-af2f67f13e95","cell_type":"code","source":"def load_slice(dcm_path):\n    ds = pydicom.dcmread(dcm_path)\n    arr = ds.pixel_array.astype(np.float32)\n    slope = float(getattr(ds, \"RescaleSlope\", 1))\n    intercept = float(getattr(ds, \"RescaleIntercept\", 0))\n    arr = arr * slope + intercept\n    return arr, ds\n\ndef middle_slice_path(series_dir):\n    files = sorted(glob.glob(os.path.join(series_dir, \"*.dcm\")))\n    return files[len(files) // 2] if files else None\n\ndemo_study = sample_studies[0]\ndemo_series_dirs = list_series_dirs(demo_study)\n\nfig, axes = plt.subplots(1, len(demo_series_dirs), figsize=(4 * len(demo_series_dirs), 4))\nif len(demo_series_dirs) == 1:\n    axes = [axes]\n\nfor ax, series_dir in zip(axes, demo_series_dirs):\n    path = middle_slice_path(series_dir)\n    if path is None:\n        continue\n    try:\n        arr, ds = load_slice(path)\n        ax.imshow(arr, cmap=\"gray\")\n        plane = getattr(ds, \"SeriesDescription\", os.path.basename(series_dir)[:12])\n        ax.set_title(plane, fontsize=8)\n    except Exception as e:\n        ax.set_title(f\"decode failed:\\n{type(e).__name__}\", fontsize=8)\n    ax.axis(\"off\")\n\nplt.suptitle(f\"Study {demo_study}\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:10:57.613581Z","iopub.execute_input":"2026-08-20T10:10:57.613949Z","iopub.status.idle":"2026-08-20T10:10:58.412861Z","shell.execute_reply.started":"2026-08-20T10:10:57.613896Z","shell.execute_reply":"2026-08-20T10:10:58.409106Z"}},"outputs":[],"execution_count":null},{"id":"342b196e-a99d-429a-b9f0-ebc253c8a0d5","cell_type":"markdown","source":"## 9. Test set sanity checks","metadata":{}},{"id":"9b4dba80-38c4-4f76-bc09-13e077481eff","cell_type":"code","source":"print(\"test.csv columns:\", list(test.columns))\nprint(\"test_series.csv columns:\", list(test_series.columns))\nprint(\"test studies:\", test[\"StudyInstanceUID\"].nunique())\nprint(\"test series total:\", len(test_series))\nprint(\"series per test study:\")\nprint(test_series.groupby(\"StudyInstanceUID\").size().describe())\n\nassert set(test[\"StudyInstanceUID\"]) == set(test_series[\"StudyInstanceUID\"])\nassert list(sample_submission.columns) == [\"StudyInstanceUID\"] + LABEL_COLS","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-20T10:11:08.366608Z","iopub.execute_input":"2026-08-20T10:11:08.367041Z","iopub.status.idle":"2026-08-20T10:11:08.378847Z","shell.execute_reply.started":"2026-08-20T10:11:08.367011Z","shell.execute_reply":"2026-08-20T10:11:08.377573Z"}},"outputs":[],"execution_count":null}]}