{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":24800,"databundleVersionId":1831594,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport os\nimport pydicom","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the root data directory\nDATA_DIR = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection\"\n\n# Define the paths to the training and testing dicom folders respectively\nTRAIN_DIR = os.path.join(DATA_DIR, \"train\")\nTEST_DIR = os.path.join(DATA_DIR, \"test\")\n\n# Capture all the relevant full train/test paths\nTRAIN_DICOM_PATHS = [os.path.join(TRAIN_DIR, f_name) for f_name in os.listdir(TRAIN_DIR)]\nTEST_DICOM_PATHS = [os.path.join(TEST_DIR, f_name) for f_name in os.listdir(TEST_DIR)]\nprint(f\"\\n... The number of training files is {len(TRAIN_DICOM_PATHS)} ...\")\nprint(f\"... The number of testing files is {len(TEST_DICOM_PATHS)} ...\")\n\n# Define paths to the relevant csv files\nTRAIN_CSV = os.path.join(DATA_DIR, \"train.csv\")\nSS_CSV = os.path.join(DATA_DIR, \"sample_submission.csv\")\n\n# Create the relevant dataframe objects\ntrain_df = pd.read_csv(TRAIN_CSV)\nss_df = pd.read_csv(SS_CSV)\n\nprint(\"\\n\\nTRAIN DATAFRAME\\n\\n\")\ndisplay(train_df.head(3))\n\nprint(\"\\n\\nSAMPLE SUBMISSION DATAFRAME\\n\\n\")\ndisplay(ss_df.head(3))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df[train_df.class_id!=14]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Useful function\ndef get_image_dimensions(dicom_paths):\n    dimensions_dict = {}\n    for path in dicom_paths:\n        dicom = pydicom.read_file(path)\n        image_id = os.path.splitext(os.path.basename(path))[0]  # Extract image ID from filename\n        dimensions_dict[image_id] = {'width': dicom.Columns, 'height': dicom.Rows}\n    return dimensions_dict","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For the train images\ntrain_original_dimensions = get_image_dimensions(TRAIN_DICOM_PATHS)\n\n# Convert the dictionary to a DataFrame\ntrain_dimensions_df = pd.DataFrame.from_dict(train_original_dimensions, orient='index').reset_index()\ntrain_dimensions_df.columns = ['image_id', 'width', 'height']\n\n# Define the path for the CSV file\ntrain_dimensions_csv_path = '/kaggle/working/train_original_dimensions.csv'\n\n# Save the DataFrame to a CSV file\ntrain_dimensions_df.to_csv(train_dimensions_csv_path, index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For the test images\ntest_original_dimensions = get_image_dimensions(TEST_DICOM_PATHS)\n\n# Convert the dictionary to a DataFrame\ntest_dimensions_df = pd.DataFrame.from_dict(test_original_dimensions, orient='index').reset_index()\ntest_dimensions_df.columns = ['image_id', 'width', 'height']\n\n# Define the path for the CSV file\ntest_dimensions_csv_path = '/kaggle/working/test_original_dimensions.csv'\n\n# Save the DataFrame to a CSV file\ntest_dimensions_df.to_csv(test_dimensions_csv_path, index=False)","metadata":{},"execution_count":null,"outputs":[]}]}