{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":113002,"databundleVersionId":13471427,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#importing relevant libraries for analysis and visualizaiton\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport os\n\n#Ensure plots display in Colab\n%matplotlib inline\n\n#set seaborn style for clean visualizations\nsns.set(style='whitegrid')\n\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-10T03:33:35.375269Z","iopub.execute_input":"2025-09-10T03:33:35.375935Z","iopub.status.idle":"2025-09-10T03:33:35.382485Z","shell.execute_reply.started":"2025-09-10T03:33:35.375911Z","shell.execute_reply":"2025-09-10T03:33:35.381546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# define path to training data\ndata_path = '/kaggle/input/grand-xray-slam-division-b/train2.csv'\n\n#load training dataset with error handling\ntry:\n    train_df = pd.read_csv(data_path)\n    print(f\"Successfully loaded train.csv with shape: {train_df.shape}\")\n\nexcept FileNotFoundError:\n    print(f\"Error: {data_path} not found. Please check the file path\")\n    raise","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T03:35:37.1735Z","iopub.execute_input":"2025-09-10T03:35:37.173949Z","iopub.status.idle":"2025-09-10T03:35:37.536488Z","shell.execute_reply.started":"2025-09-10T03:35:37.173915Z","shell.execute_reply":"2025-09-10T03:35:37.535574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T03:35:49.744258Z","iopub.execute_input":"2025-09-10T03:35:49.744907Z","iopub.status.idle":"2025-09-10T03:35:49.76031Z","shell.execute_reply.started":"2025-09-10T03:35:49.74488Z","shell.execute_reply":"2025-09-10T03:35:49.759523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Dataset Info: \")\nprint(train_df.info())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T03:37:17.792365Z","iopub.execute_input":"2025-09-10T03:37:17.792704Z","iopub.status.idle":"2025-09-10T03:37:17.839252Z","shell.execute_reply.started":"2025-09-10T03:37:17.792678Z","shell.execute_reply":"2025-09-10T03:37:17.838327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#summary key metrics\ntotal_images = len(train_df)\ntotal_patients = train_df['Patient_ID'].nunique()\ntotal_studies= train_df['Study'].nunique()\nprint(f\"Total Images: {total_images}\")\nprint(f\"Total Patients: {total_patients}\")\nprint(f\"Total Studies: {total_studies}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T03:41:26.418953Z","iopub.execute_input":"2025-09-10T03:41:26.419652Z","iopub.status.idle":"2025-09-10T03:41:26.430617Z","shell.execute_reply.started":"2025-09-10T03:41:26.419591Z","shell.execute_reply":"2025-09-10T03:41:26.429709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Check for missing values\nprint(\"\\nMissing Values: \")\nprint(train_df.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T03:42:36.399648Z","iopub.execute_input":"2025-09-10T03:42:36.399942Z","iopub.status.idle":"2025-09-10T03:42:36.431731Z","shell.execute_reply.started":"2025-09-10T03:42:36.39992Z","shell.execute_reply":"2025-09-10T03:42:36.430359Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Label Prevalence Analysis\n\nAnalyze the distribution of the 14 conditions to identify class imbalance.","metadata":{}},{"cell_type":"code","source":"#Define the 14 condition columns\nlabel_columns = ['No Finding', 'Lung Opacity', 'Support Devices', 'Atelectasis', \n                'Cardiomegaly', 'Pleural Effusion', 'Enlarged Cardiomediastinum',\n                'Edema', 'Consolidation', 'Pneumonia', 'Fracture', 'Lung Lesion',\n                'Pneumothorax', 'Pleural Other']\n\n#calculate counts and percentages for each condition\nlabel_counts = train_df[label_columns].sum()\nlabel_percentages = (label_counts / total_images * 100).round(2)\nprevalence_df = pd.DataFrame({\n    'Condition': label_counts.index,\n    'Count': label_counts.values,\n    'Percent (%)': label_percentages.values\n})\n\n#display prevalence table\nprint(\"label prevalence\")\nprint(prevalence_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T03:49:45.338218Z","iopub.execute_input":"2025-09-10T03:49:45.338539Z","iopub.status.idle":"2025-09-10T03:49:45.352912Z","shell.execute_reply.started":"2025-09-10T03:49:45.338519Z","shell.execute_reply":"2025-09-10T03:49:45.35187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nsns.barplot(x='Count', y='Condition', data=prevalence_df, palette='viridis', hue=None )\nplt.title('Label counts (Number of Positive Cases)')\nplt.xlabel('Count')\nplt.ylabel('Condition')\nplt.legend([],[], frameon=False)\nplt.tight_layout()\nplt.savefig('/content/label_counts_barplot.jpg')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T03:54:33.51262Z","iopub.execute_input":"2025-09-10T03:54:33.512935Z","iopub.status.idle":"2025-09-10T03:54:34.290434Z","shell.execute_reply.started":"2025-09-10T03:54:33.512914Z","shell.execute_reply":"2025-09-10T03:54:34.289536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# donut chart for label percentages\nplt.figure(figsize=(8,8))\ncolors = sns.color_palette('viridis', len(prevalence_df))\nplt.pie(prevalence_df['Count'], labels=prevalence_df['Condition'],\n       autopct=lambda pct: f'{pct:.1f}%', )","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}