{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":112899,"databundleVersionId":13449579,"sourceType":"competition"},{"sourceId":260372162,"sourceType":"kernelVersion"},{"sourceId":265365994,"sourceType":"kernelVersion"},{"sourceId":265982222,"sourceType":"kernelVersion"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nfrom scipy.special import logit, expit\nimport matplotlib.pyplot as plt\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T07:33:38.756133Z","iopub.execute_input":"2025-10-03T07:33:38.757238Z","iopub.status.idle":"2025-10-03T07:33:38.761968Z","shell.execute_reply.started":"2025-10-03T07:33:38.757201Z","shell.execute_reply":"2025-10-03T07:33:38.76096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/grand-xray-slam-division-a/train1.csv')\ntargets = train.columns[-14:]\npriors = train[targets].mean(axis=0)\nprint(priors)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T07:33:38.763269Z","iopub.execute_input":"2025-10-03T07:33:38.763577Z","iopub.status.idle":"2025-10-03T07:33:39.123613Z","shell.execute_reply.started":"2025-10-03T07:33:38.763547Z","shell.execute_reply":"2025-10-03T07:33:39.122666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"eff_sub=pd.read_csv(\"/kaggle/input/gxsa-efficientnetv2-b0-352x352/submission.csv\")\nind_sub=pd.read_csv(\"/kaggle/input/gxsa-individual/submission.csv\")\nseres_sub=pd.read_csv(\"/kaggle/input/xraydivisiona-groupedstudylevel/submission_study_level_avg.csv\")\n\n# Balance the classes in ind_sub\nind_sub[targets] = expit(logit(ind_sub[targets]) + logit(priors))\n\n# Bugfix for eff_sub\neff_sub['Patient_Study'] = eff_sub['Image_name'].str.slice(0, 12).astype(int)\neff_sub[targets] = eff_sub[targets].groupby(eff_sub['Patient_Study'].values).transform(lambda x: x.values.mean()).values\n\n# Sort all by Image_name\neff_sub = eff_sub.sort_values(by=\"Image_name\").reset_index(drop=True)\nind_sub = ind_sub.sort_values(by=\"Image_name\").reset_index(drop=True)\nseres_sub = seres_sub.sort_values(by=\"Image_name\").reset_index(drop=True)\n\n# Verify sorting\nprint((eff_sub.Image_name == seres_sub.Image_name).mean())\nprint((ind_sub.Image_name == seres_sub.Image_name).mean())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T07:33:39.125117Z","iopub.execute_input":"2025-10-03T07:33:39.125497Z","iopub.status.idle":"2025-10-03T07:34:07.57291Z","shell.execute_reply.started":"2025-10-03T07:33:39.125464Z","shell.execute_reply":"2025-10-03T07:34:07.572073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"w_ind = 0.25\nw_eff = 0.45\nw_seres = 0.30\n\n# Weighted average directly\nensemble_df = eff_sub[['Image_name']].copy()\nfor col in targets:\n    plt.figure(figsize=(12, 4))\n    plt.subplot(1, 3, 1)\n    plt.scatter(ind_sub[col], seres_sub[col], s=1)\n    plt.gca().set_aspect('equal')\n    plt.subplot(1, 3, 2)\n    plt.scatter(ind_sub[col], eff_sub[col], s=1)\n    plt.gca().set_aspect('equal')\n    plt.subplot(1, 3, 3)\n    plt.scatter(eff_sub[col], seres_sub[col], s=1)\n    plt.gca().set_aspect('equal')\n    plt.show()\n    ensemble_df[col] = (w_ind * ind_sub[col] + w_eff * eff_sub[col] + w_seres * seres_sub[col]) / (w_ind + w_eff + w_seres)\n\n# Save\nensemble_df.to_csv(\"ensemble_predictions.csv\", index=False)\nprint(\"✅ Ensemble saved to ensemble_predictions.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T07:34:07.573805Z","iopub.execute_input":"2025-10-03T07:34:07.574092Z","iopub.status.idle":"2025-10-03T07:34:14.802297Z","shell.execute_reply.started":"2025-10-03T07:34:07.57407Z","shell.execute_reply":"2025-10-03T07:34:14.801463Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ensemble_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T07:34:14.803653Z","iopub.execute_input":"2025-10-03T07:34:14.803911Z","iopub.status.idle":"2025-10-03T07:34:14.822704Z","shell.execute_reply.started":"2025-10-03T07:34:14.80389Z","shell.execute_reply":"2025-10-03T07:34:14.821823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}