{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\nimport os\nfrom pathlib import Path\nimport pydicom\n\n# 1. Define base data directory\nINPUT_DIR = Path('/kaggle/input/competitions/rsna-knee-abnormality-detection')\n\n# 2. Load the metadata CSV tables\ntrain_df = pd.read_csv(INPUT_DIR / 'train.csv')\ntrain_series_df = pd.read_csv(INPUT_DIR / 'train_series.csv')\ntest_df = pd.read_csv(INPUT_DIR / 'test.csv')\ntest_series_df = pd.read_csv(INPUT_DIR / 'test_series.csv')\nsample_sub = pd.read_csv(INPUT_DIR / 'sample_submission.csv')\n\n# 3. Print summaries\nprint(f\"Train studies: {len(train_df)}\")\nprint(f\"Train series: {len(train_series_df)}\")\nprint(f\"Test studies (Public Sample): {len(test_df)}\")\nprint(f\"Sample submission shape: {sample_sub.shape}\")\n\n# 4. Preview train.csv structure\ntrain_df.head(10)\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-08-22T06:50:38.142153Z","iopub.status.idle":"2026-08-22T06:50:38.142374Z","shell.execute_reply.started":"2026-08-22T06:50:38.142267Z","shell.execute_reply":"2026-08-22T06:50:38.14228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T06:50:38.143804Z","iopub.status.idle":"2026-08-22T06:50:38.144175Z","shell.execute_reply.started":"2026-08-22T06:50:38.14397Z","shell.execute_reply":"2026-08-22T06:50:38.144005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Check total rows vs labeled rows\ntotal_studies = len(train_df)\nlabeled_mask = train_df['ACL'].notna()\ngold_count = labeled_mask.sum()\nunlabeled_count = total_studies - gold_count\n\nprint(f\"Total Studies: {total_studies}\")\nprint(f\"Gold-labeled studies (with ground truth): {gold_count} ({gold_count / total_studies * 100:.2f}%)\")\nprint(f\"Weakly-supervised studies (NaNs, reports only): {unlabeled_count} ({unlabeled_count / total_studies * 100:.2f}%)\")\n\n# 2. View the gold-labeled subset\ntrain_df[labeled_mask].head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T06:50:38.145247Z","iopub.status.idle":"2026-08-22T06:50:38.145549Z","shell.execute_reply.started":"2026-08-22T06:50:38.145359Z","shell.execute_reply":"2026-08-22T06:50:38.145372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport glob \nimport matplotlib.pyplot as plt\n\n# 1. Define paths\nTRAIN_DIR = INPUT_DIR / 'train_series'\n\n# 2. Load CSV metadata\ntrain_df = pd.read_csv(INPUT_DIR / 'train.csv')\ntrain_series_df = pd.read_csv(INPUT_DIR / 'train_series.csv')\n\n# 3. Choose a specific study to visualize (e.g., first study with gold labels or any UID)\ngold_studies = train_df[train_df['ACL'].notna()]['StudyInstanceUID'].values\nstudy_id = gold_studies[0] if len(gold_studies) > 0 else train_df['StudyInstanceUID'].iloc[0]\n\nstudy_meta = train_df[train_df['StudyInstanceUID'] == study_id].iloc[0]\nseries_rows = train_series_df[train_series_df['StudyInstanceUID'] == study_id]\n\n# 4. Print metadata & report\nprint(\"=\" * 80)\nprint(f\"StudyInstanceUID: {study_id}\")\nprint(\"=\" * 80)\nprint(\"\\n--- CLINICAL LABELS ---\")\ntarget_cols = ['ACL', 'MCL', 'Medial Meniscus', 'Lateral Meniscus', 'Medial OA', \n               'Lateral OA', 'PF OA', 'Effusion', 'Synovitis', \"Baker's\", 'Contusion', 'Fracture']\nfor col in target_cols:\n    val = study_meta[col] if col in study_meta else \"N/A\"\n    print(f\"  {col:18s}: {val}\")\n\nprint(\"\\n--- RADIOLOGY REPORT ---\")\nprint(str(study_meta.get('Report', 'No report text available.'))[:500] + \"...\\n\")\n\n# 5. Helper function: Sort DICOM slices physically in 3D space\ndef load_sorted_series(series_dir):\n    dcm_files = glob.glob(os.path.join(series_dir, '*.dcm'))\n    if not dcm_files:\n        return np.array([])\n    \n    # Read headers to sort along slice normal vector\n    slice_data = []\n    for fpath in dcm_files:\n        try:\n            d = pydicom.dcmread(fpath, stop_before_pixels=False)\n            pos = float(d.ImagePositionPatient[2]) if hasattr(d, 'ImagePositionPatient') else 0.0\n            slice_data.append((pos, d.pixel_array))\n        except Exception:\n            continue\n            \n    # Sort by physical Z-position\n    slice_data.sort(key=lambda x: x[0])\n    return np.array([arr for _, arr in slice_data])\n\n# 6. Plot available MRI Series (Sagittal, Coronal, Axial)\nnum_series = len(series_rows)\nprint(f\"Total MRI Series for this Study: {num_series}\")\n\nfig, axes = plt.subplots(num_series, 4, figsize=(16, 4 * num_series))\nif num_series == 1:\n    axes = np.expand_dims(axes, 0)\n\nfor row_idx, (_, s_row) in enumerate(series_rows.iterrows()):\n    s_uid = s_row['SeriesInstanceUID']\n    plane = s_row.get('Anatomical_Plane', 'Unknown Plane')\n    fluid_sens = \"Fluid-Sens\" if s_row.get('Fluid_Sensitive', 0) == 1 else \"Non-Fluid\"\n    fat_supp = \"Fat-Supp\" if s_row.get('Fat_Suppression', 0) == 1 else \"No Fat-Supp\"\n    \n    series_path = TRAIN_DIR / study_id / s_uid\n    volume = load_sorted_series(series_path)\n    \n    if len(volume) == 0:\n        continue\n        \n    # Sample 4 slices uniformly across the 3D volume\n    slice_indices = np.linspace(0, len(volume) - 1, 4, dtype=int)\n    \n    for col_idx, s_idx in enumerate(slice_indices):\n        ax = axes[row_idx, col_idx]\n        ax.imshow(volume[s_idx], cmap='bone')\n        ax.axis('off')\n        \n        if col_idx == 0:\n            ax.set_title(f\"{plane} ({fluid_sens}, {fat_supp})\\nSlice {s_idx + 1}/{len(volume)}\", fontsize=10, loc='left')\n        else:\n            ax.set_title(f\"Slice {s_idx + 1}/{len(volume)}\", fontsize=10)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T06:52:01.867906Z","iopub.execute_input":"2026-08-22T06:52:01.868169Z","iopub.status.idle":"2026-08-22T06:52:05.671064Z","shell.execute_reply.started":"2026-08-22T06:52:01.868147Z","shell.execute_reply":"2026-08-22T06:52:05.669943Z"}},"outputs":[],"execution_count":null}]}