{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# RSNA Knee Abnormality Detection — Data Audit\n\n**Run this notebook top-to-bottom on Kaggle before writing any modeling code.**\nEvery cell below is written against RSNA's standard multi-competition DICOM\nconvention (`[train/test]_images/{StudyInstanceUID}/{SeriesInstanceUID}/{instance}.dcm`),\nbut this is inferred from other RSNA competitions, not confirmed against this\ncompetition's actual mounted files. Cell 1 verifies it before anything else\nassumes it's correct — same lesson as the model-path issue on the other\ncompetition: confirm the real path before building on top of a guess.\n","metadata":{}},{"cell_type":"markdown","source":"## Step 1 — confirm the actual file structure (confirmed, not guessed)\n\nStructure is: `train_series/{StudyInstanceUID}/{SeriesInstanceUID}/{instance}.dcm`\n— same shape as other RSNA competitions, just the top folder is named `train_series`\n(matching `train_series.csv`) instead of `train_images`. Confirmed by listing real\nfolders below rather than assuming.","metadata":{}},{"cell_type":"code","source":"import os\n\nCOMPETITION_INPUT = '/kaggle/input/competitions/rsna-knee-abnormality-detection'\ntrain_series_dir = os.path.join(COMPETITION_INPUT, 'train_series')\n\nstudy_dirs = sorted(os.listdir(train_series_dir))\nprint(f\"train_series/ contains {len(study_dirs)} study folders\")\n\nfirst_study = study_dirs[0]\nseries_dirs = sorted(os.listdir(os.path.join(train_series_dir, first_study)))\nprint(f\"train_series/{first_study}/ contains {len(series_dirs)} series folders\")\n\nfirst_series = series_dirs[0]\nfiles = sorted(os.listdir(os.path.join(train_series_dir, first_study, first_series)))\nprint(f\"train_series/{first_study}/{first_series}/ contains {len(files)} files\")\nprint(\"First few:\", files[:3])\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 2 — read one DICOM file, inspect the header tags that matter","metadata":{}},{"cell_type":"code","source":"import pydicom\n\nsample_dcm_path = os.path.join(train_series_dir, first_study, first_series, files[0])\nds = pydicom.dcmread(sample_dcm_path)\n\nprint(\"Key header tags:\")\nfor tag in ['Rows', 'Columns', 'PixelSpacing', 'SliceThickness', 'SpacingBetweenSlices',\n            'ImageOrientationPatient', 'ImagePositionPatient', 'InstanceNumber',\n            'PhotometricInterpretation', 'BitsStored', 'RescaleSlope', 'RescaleIntercept',\n            'SeriesDescription', 'Laterality', 'BodyPartExamined']:\n    val = getattr(ds, tag, '<not present>')\n    print(f\"  {tag}: {val}\")\n\nprint()\nprint(\"NOTE: check the 'Laterality' line above. If it's <not present>, laterality\")\nprint(\"(left/right knee) will need to come from somewhere else — SeriesDescription,\")\nprint(\"the Report text, or possibly it's just not distinguished in this dataset.\")\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 3 — audit a sample of studies: dimensions, spacing, slice counts, laterality","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport random\n\nrandom.seed(42)\nsample_studies = random.sample(study_dirs, min(40, len(study_dirs)))\n\naudit_rows = []\nfor study in sample_studies:\n    study_path = os.path.join(train_series_dir, study)\n    for series in os.listdir(study_path):\n        series_path = os.path.join(study_path, series)\n        dcm_files = sorted(os.listdir(series_path))\n        if not dcm_files:\n            continue\n        # read just the first slice's header per series — fast, avoids full pixel decode\n        ds = pydicom.dcmread(os.path.join(series_path, dcm_files[0]), stop_before_pixels=True)\n        audit_rows.append({\n            'StudyInstanceUID': study,\n            'SeriesInstanceUID': series,\n            'n_slices': len(dcm_files),\n            'rows': getattr(ds, 'Rows', None),\n            'cols': getattr(ds, 'Columns', None),\n            'pixel_spacing': str(getattr(ds, 'PixelSpacing', None)),\n            'slice_thickness': getattr(ds, 'SliceThickness', None),\n            'laterality': getattr(ds, 'Laterality', None),\n            'series_description': getattr(ds, 'SeriesDescription', None),\n        })\n\naudit_df = pd.DataFrame(audit_rows)\nprint(f\"Audited {len(sample_studies)} studies, {len(audit_df)} series\\n\")\nprint(\"Slices per series:\")\nprint(audit_df['n_slices'].describe())\nprint()\nprint(\"Image dimensions (rows x cols) value counts:\")\nprint((audit_df['rows'].astype(str) + ' x ' + audit_df['cols'].astype(str)).value_counts())\nprint()\nprint(\"Laterality tag presence:\", audit_df['laterality'].notna().sum(), \"/\", len(audit_df))\nif audit_df['laterality'].notna().any():\n    print(audit_df['laterality'].value_counts())\nprint()\nprint(\"Sample of series descriptions (useful for confirming plane/sequence type):\")\nprint(audit_df['series_description'].dropna().unique()[:20])\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 4 — cross-check against `train_series.csv`","metadata":{}},{"cell_type":"code","source":"train_series = pd.read_csv(os.path.join(COMPETITION_INPUT, 'train_series.csv'))\n\nmerged = audit_df.merge(\n    train_series[['StudyInstanceUID', 'SeriesInstanceUID', 'Anatomical_Plane']],\n    on=['StudyInstanceUID', 'SeriesInstanceUID'], how='left'\n)\nprint(\"Series found on disk but missing from train_series.csv:\", merged['Anatomical_Plane'].isna().sum())\nprint()\nprint(\"Slice count by anatomical plane (sanity check — planes shouldn't have wildly different slice counts):\")\nprint(merged.groupby('Anatomical_Plane')['n_slices'].describe())\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 5 — visualize a few slices\n\nQuick visual sanity check — confirms images actually look like knee MRIs, orientation\nlooks sane, and there's no scaling/inversion issue before building a training pipeline\non top of this.","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\nfig, axes = plt.subplots(1, 3, figsize=(13, 4.5))\nshown = 0\nfor study in sample_studies:\n    if shown >= 3:\n        break\n    study_path = os.path.join(train_series_dir, study)\n    series_list = os.listdir(study_path)\n    if not series_list:\n        continue\n    series_path = os.path.join(study_path, series_list[0])\n    dcm_files = sorted(os.listdir(series_path))\n    if not dcm_files:\n        continue\n    mid_file = dcm_files[len(dcm_files) // 2]  # middle slice, usually most representative\n    ds = pydicom.dcmread(os.path.join(series_path, mid_file))\n    img = ds.pixel_array.astype(float)\n    # apply rescale if present\n    slope = float(getattr(ds, 'RescaleSlope', 1))\n    intercept = float(getattr(ds, 'RescaleIntercept', 0))\n    img = img * slope + intercept\n    axes[shown].imshow(img, cmap='gray')\n    axes[shown].set_title(f\"{study[-8:]}...\\n{getattr(ds, 'SeriesDescription', '')[:25]}\", fontsize=9)\n    axes[shown].axis('off')\n    shown += 1\n\nplt.tight_layout()\nplt.show()\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Laterality resolution — full corpus\n\nRun this on Kaggle after the data-audit notebook. Reads one DICOM header per series\n(fast — header-only, `stop_before_pixels=True`) across the whole `train_series.csv`,\nresolving laterality via:\n\n1. The DICOM `Laterality` tag, when non-empty.\n2. A regex fallback on `SeriesDescription` (R/L prefixes like `'R SAG T1 FSE'`,\n   `'R- CORPD FSE FS'`) when the tag is missing or blank.\n3. Otherwise marked unresolved.\n\nDefensive against garbage `SeriesDescription` values (saw a literal `'DummySeriesDesc!'`\nin the audit sample) — the regex simply won't match junk, which is the correct behavior.","metadata":{}},{"cell_type":"code","source":"import os\nimport re\nimport pydicom\nimport pandas as pd\nfrom tqdm import tqdm\n\nCOMPETITION_INPUT = '/kaggle/input/competitions/rsna-knee-abnormality-detection'\ntrain_series_dir = os.path.join(COMPETITION_INPUT, 'train_series')\ntrain_series_csv = pd.read_csv(os.path.join(COMPETITION_INPUT, 'train_series.csv'))\n\n# Matches an R/L token at the start of the description, or as a standalone word/prefix.\n# Deliberately conservative — false \"resolved\" is worse than leaving something unresolved.\nLATERALITY_PATTERN = re.compile(r'^(R|L|R-|L-|Right|Left)\\b', re.IGNORECASE)\n\ndef resolve_laterality(tag_value, series_description):\n    if tag_value and str(tag_value).strip():\n        return str(tag_value).strip().upper()[0], 'tag'\n    if series_description:\n        m = LATERALITY_PATTERN.match(str(series_description).strip())\n        if m:\n            return m.group(1)[0].upper(), 'description'\n    return None, 'unresolved'\n\nrows = []\nfor _, r in tqdm(train_series_csv.iterrows(), total=len(train_series_csv)):\n    study, series = r['StudyInstanceUID'], r['SeriesInstanceUID']\n    series_path = os.path.join(train_series_dir, study, series)\n    try:\n        dcm_files = os.listdir(series_path)\n        if not dcm_files:\n            continue\n        ds = pydicom.dcmread(os.path.join(series_path, dcm_files[0]), stop_before_pixels=True)\n    except Exception as e:\n        rows.append({'StudyInstanceUID': study, 'SeriesInstanceUID': series,\n                      'laterality': None, 'source': f'read_error: {e}'})\n        continue\n\n    tag_value = getattr(ds, 'Laterality', None)\n    series_desc = getattr(ds, 'SeriesDescription', None)\n    laterality, source = resolve_laterality(tag_value, series_desc)\n    rows.append({'StudyInstanceUID': study, 'SeriesInstanceUID': series,\n                  'laterality': laterality, 'source': source})\n\nlaterality_df = pd.DataFrame(rows)\nlaterality_df.to_csv('/kaggle/working/laterality_lookup.csv', index=False)\n\nprint(f\"Resolved {len(laterality_df)} series\\n\")\nprint(\"Resolution source breakdown:\")\nprint(laterality_df['source'].value_counts())\nprint()\nprint(\"Laterality value breakdown (where resolved):\")\nprint(laterality_df['laterality'].value_counts(dropna=False))\nprint()\ncoverage = (laterality_df['laterality'].notna()).mean() * 100\nprint(f\"Overall laterality coverage: {coverage:.1f}%\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub = pd.read_csv(os.path.join(COMPETITION_INPUT, 'sample_submission.csv'))\nprint(sample_sub.columns.tolist())\nprint(sample_sub.head())\n\ntrain_csv = pd.read_csv(os.path.join(COMPETITION_INPUT, 'train.csv'))  # or whatever the labels file is called\nprint(train_csv.columns.tolist())\nprint(train_csv.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T17:51:21.043887Z","iopub.execute_input":"2026-09-16T17:51:21.044191Z","iopub.status.idle":"2026-09-16T17:51:21.065441Z","shell.execute_reply.started":"2026-09-16T17:51:21.04416Z","shell.execute_reply":"2026-09-16T17:51:21.063962Z"}},"outputs":[],"execution_count":null}]}