{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":99552,"databundleVersionId":13851420,"sourceType":"competition"},{"sourceId":266169724,"sourceType":"kernelVersion"}],"dockerImageVersionId":31153,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-06T18:42:33.409972Z","iopub.execute_input":"2025-10-06T18:42:33.410278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\nMEDICAL INFERENCE - Matches guaranteed training\nFull medical features for submission\n\"\"\"\n\nimport os\nimport shutil\nfrom collections import defaultdict\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport pydicom\nimport pickle\nimport gc\n\nimport kaggle_evaluation.rsna_inference_server\n\nID_COL = 'SeriesInstanceUID'\n\nLABEL_COLS = [\n    'Left Infraclinoid Internal Carotid Artery',\n    'Right Infraclinoid Internal Carotid Artery',\n    'Left Supraclinoid Internal Carotid Artery',\n    'Right Supraclinoid Internal Carotid Artery',\n    'Left Middle Cerebral Artery',\n    'Right Middle Cerebral Artery',\n    'Anterior Communicating Artery',\n    'Left Anterior Cerebral Artery',\n    'Right Anterior Cerebral Artery',\n    'Left Posterior Communicating Artery',\n    'Right Posterior Communicating Artery',\n    'Basilar Tip',\n    'Other Posterior Circulation',\n    'Aneurysm Present',\n]\n\nclass MedicalInference:\n    def __init__(self):\n        self.models = None\n        self.feature_cols = None\n        self.prevalence = None\n        self._load_models()\n    \n    def _load_models(self):\n        \"\"\"Load medical models.\"\"\"\n        \n        paths = [\n            'medical_model.pkl',\n            'medical_model_v2.pkl',\n            '/kaggle/input/rsna-medical-aneurysm/medical_model.pkl',\n            '/kaggle/input/rsna-medical-aneurysm/medical_model_v2.pkl',\n        ]\n        \n        for path in paths:\n            if os.path.exists(path):\n                try:\n                    with open(path, 'rb') as f:\n                        data = pickle.load(f)\n                        self.models = data['models']\n                        self.feature_cols = data['feature_cols']\n                        self.prevalence = data.get('prevalence', {})\n                    print(f\"✅ Models loaded: {path}\")\n                    print(f\"   Features: {len(self.feature_cols)}\")\n                    print(f\"   Models: {len(self.models)}\")\n                    return\n                except Exception as e:\n                    print(f\"⚠️  Failed to load {path}: {e}\")\n        \n        print(\"⚠️  No models found, using medical priors\")\n        self._set_priors()\n    \n    def _set_priors(self):\n        \"\"\"Medical priors from literature.\"\"\"\n        self.prevalence = {\n            'Anterior Communicating Artery': 0.325,\n            'Left Posterior Communicating Artery': 0.125,\n            'Right Posterior Communicating Artery': 0.125,\n            'Left Middle Cerebral Artery': 0.10,\n            'Right Middle Cerebral Artery': 0.10,\n            'Basilar Tip': 0.06,\n            'Left Supraclinoid Internal Carotid Artery': 0.04,\n            'Right Supraclinoid Internal Carotid Artery': 0.04,\n            'Other Posterior Circulation': 0.05,\n            'Left Anterior Cerebral Artery': 0.025,\n            'Right Anterior Cerebral Artery': 0.025,\n            'Left Infraclinoid Internal Carotid Artery': 0.015,\n            'Right Infraclinoid Internal Carotid Artery': 0.015,\n            'Aneurysm Present': 0.43,\n        }\n        self.feature_cols = []\n    \n    def extract_features(self, series_path, tags):\n        \"\"\"Extract same medical features.\"\"\"\n        \n        features = {}\n        filepaths = tags.get('filepath', [])\n        \n        if not filepaths:\n            return self._defaults()\n        \n        try:\n            ds = pydicom.dcmread(filepaths[0], force=True)\n            \n            # Modality\n            modality = str(getattr(ds, 'Modality', 'Unknown'))\n            features['is_ct'] = 1.0 if modality == 'CT' else 0.0\n            features['is_mr'] = 1.0 if modality == 'MR' else 0.0\n            \n            # Protocol\n            features['num_slices'] = float(len(filepaths))\n            features['num_slices_log'] = np.log(len(filepaths) + 1)\n            \n            slice_thickness = float(getattr(ds, 'SliceThickness', 1.0))\n            features['slice_thickness'] = slice_thickness\n            features['is_thin'] = 1.0 if slice_thickness < 0.7 else 0.0\n            features['is_good'] = 1.0 if slice_thickness < 1.0 else 0.0\n            \n            pixel_spacing = getattr(ds, 'PixelSpacing', [0.5, 0.5])\n            features['pixel_spacing'] = float(pixel_spacing[0])\n            features['is_high_res'] = 1.0 if float(pixel_spacing[0]) < 0.4 else 0.0\n            \n            features['protocol_quality'] = (\n                (1.0 if features['is_thin'] else 0.5) *\n                (1.0 if features['is_high_res'] else 0.7) *\n                (1.0 if features['num_slices'] > 100 else 0.8)\n            )\n            \n            # Intensity\n            n_slices = len(filepaths)\n            sample_indices = [\n                int(n_slices * 0.3),\n                int(n_slices * 0.5),\n                int(n_slices * 0.7),\n            ]\n            \n            intensities = []\n            for idx in sample_indices:\n                if 0 <= idx < n_slices:\n                    ds_sample = pydicom.dcmread(filepaths[idx], force=True)\n                    if hasattr(ds_sample, 'pixel_array'):\n                        pixels = ds_sample.pixel_array.astype(float)\n                        intercept = getattr(ds_sample, 'RescaleIntercept', 0)\n                        slope = getattr(ds_sample, 'RescaleSlope', 1)\n                        pixels = pixels * slope + intercept\n                        intensities.append(pixels)\n            \n            if intensities:\n                all_pixels = np.concatenate([img.flatten() for img in intensities])\n                features['intensity_mean'] = np.mean(all_pixels)\n                features['intensity_std'] = np.std(all_pixels)\n                features['intensity_max'] = np.max(all_pixels)\n                features['intensity_p95'] = np.percentile(all_pixels, 95)\n                features['intensity_p99'] = np.percentile(all_pixels, 99)\n                features['enhancement_ratio'] = features['intensity_p99'] / (features['intensity_mean'] + 1.0)\n                features['contrast_quality'] = (features['intensity_p95'] - features['intensity_mean']) / (features['intensity_mean'] + 1.0)\n                features['has_strong_enhancement'] = 1.0 if features['enhancement_ratio'] > 3.0 else 0.0\n            else:\n                features.update({\n                    'intensity_mean': 0.0, 'intensity_std': 0.0, 'intensity_max': 0.0,\n                    'intensity_p95': 0.0, 'intensity_p99': 0.0,\n                    'enhancement_ratio': 1.0, 'contrast_quality': 0.0,\n                    'has_strong_enhancement': 0.0,\n                })\n            \n            # Interactions\n            features['ct_with_enhancement'] = features['is_ct'] * features['enhancement_ratio']\n            features['quality_coverage'] = features['protocol_quality'] * features['num_slices_log']\n            features['optimal_for_small'] = features['is_thin'] * features['has_strong_enhancement']\n            \n        except:\n            return self._defaults()\n        \n        return features\n    \n    def _defaults(self):\n        return {\n            'is_ct': 0.0, 'is_mr': 0.0, 'num_slices': 100.0, 'num_slices_log': 4.6,\n            'slice_thickness': 1.0, 'is_thin': 0.0, 'is_good': 0.0,\n            'pixel_spacing': 0.5, 'is_high_res': 0.0, 'protocol_quality': 0.5,\n            'intensity_mean': 0.0, 'intensity_std': 0.0, 'intensity_max': 0.0,\n            'intensity_p95': 0.0, 'intensity_p99': 0.0,\n            'enhancement_ratio': 1.0, 'contrast_quality': 0.0,\n            'has_strong_enhancement': 0.0,\n            'ct_with_enhancement': 0.0, 'quality_coverage': 0.0,\n            'optimal_for_small': 0.0,\n        }\n    \n    def predict(self, series_path, tags):\n        \"\"\"Predict with medical model.\"\"\"\n        \n        features = self.extract_features(series_path, tags)\n        \n        if not self.models:\n            return {col: self.prevalence.get(col, 0.35) for col in LABEL_COLS}\n        \n        predictions = {}\n        feature_vector = np.array([[features.get(col, 0.0) for col in self.feature_cols]])\n        \n        for target in LABEL_COLS:\n            if target not in self.models:\n                predictions[target] = self.prevalence.get(target, 0.35)\n                continue\n            \n            fold_preds = []\n            for fold_model in self.models[target]:\n                scaler = fold_model['scaler']\n                model = fold_model['model']\n                \n                X_scaled = scaler.transform(feature_vector)\n                pred = model.predict_proba(X_scaled)[0, 1]\n                fold_preds.append(pred)\n            \n            predictions[target] = np.clip(np.mean(fold_preds), 0.01, 0.99)\n        \n        return predictions\n\ndef predict(series_path: str) -> pl.DataFrame:\n    \"\"\"Main predict function.\"\"\"\n    \n    if not hasattr(predict, 'predictor'):\n        predict.predictor = MedicalInference()\n    \n    predictor = predict.predictor\n    series_id = os.path.basename(series_path)\n    \n    # Collect files\n    all_filepaths = []\n    for root, _, files in os.walk(series_path):\n        for file in files:\n            if file.endswith('.dcm'):\n                all_filepaths.append(os.path.join(root, file))\n    all_filepaths.sort()\n    \n    tags = {'filepath': all_filepaths}\n    \n    # Predict\n    predictions_dict = predictor.predict(series_path, tags)\n    \n    # Format\n    row = [series_id] + [predictions_dict[col] for col in LABEL_COLS]\n    predictions = pl.DataFrame(data=[row], schema=[ID_COL, *LABEL_COLS], orient='row')\n    \n    # Cleanup\n    gc.collect()\n    shutil.rmtree('/kaggle/shared', ignore_errors=True)\n    \n    return predictions.drop(ID_COL)\n\n# Server\ninference_server = kaggle_evaluation.rsna_inference_server.RSNAInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    print(\"COMPETITION MODE - Medical Model\")\n    inference_server.serve()\nelse:\n    print(\"TEST MODE - Medical Model\")\n    inference_server.run_local_gateway()\n    display(pl.read_parquet('/kaggle/working/submission.parquet'))","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}