{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install pydicom","metadata":{"execution":{"iopub.status.busy":"2023-02-26T15:12:34.157121Z","iopub.execute_input":"2023-02-26T15:12:34.157566Z","iopub.status.idle":"2023-02-26T15:12:45.86127Z","shell.execute_reply.started":"2023-02-26T15:12:34.157524Z","shell.execute_reply":"2023-02-26T15:12:45.859891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pydicom\nimport csv\nfrom tqdm import tqdm\n\n# Define the directory paths containing the DICOM files\ntrain_dir = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train\"\ntest_dir = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/test\"\n\n# Define the list to store the metadata\nmetadata_list = []\n\n# Define the batch size\nbatch_size = 100\n\n# Iterate over each directory of DICOM files\nfor directory in [train_dir, test_dir]:\n    # Get the list of file names in the directory\n    file_names = os.listdir(directory)\n    # Sort the file names in ascending order\n    file_names.sort()\n    # Initialize a progress bar\n    progress_bar = tqdm(total=len(file_names), desc=f\"Processing {directory} files\")\n    # Define the output file name\n    output_file = f\"{directory.split('/')[-1]}_metadata.csv\"\n    output_path = os.path.join('/kaggle/working', output_file)\n    # Iterate over each file in the directory\n    for file_name in file_names:\n        # Skip non-DICOM files\n        if not file_name.endswith(\".dicom\"):\n            continue\n        # Load the DICOM file\n        file_path = os.path.join(directory, file_name)\n        ds = pydicom.dcmread(file_path)\n        # Retrieve the patient's age and gender\n        if hasattr(ds, \"PatientAge\"):\n            patient_age = int(ds.PatientAge[:-1]) if ds.PatientAge[:-1].isdigit() else \"N/A\"\n        else:\n            patient_age = \"N/A\"\n        patient_gender = ds.PatientSex if ds.PatientSex in [\"M\", \"F\"] else \"N/A\"\n        # Append the metadata to the list\n        metadata_list.append([file_name[:-6], patient_age, patient_gender])\n        # Check if we've accumulated enough metadata to write a batch to the CSV file\n        if len(metadata_list) == batch_size:\n            # Write the batch of metadata to the CSV file\n            with open(output_path, 'a', newline='') as file:\n                writer = csv.writer(file)\n                for metadata in metadata_list:\n                    writer.writerow(metadata)\n            # Clear the metadata list\n            metadata_list = []\n        # Update the progress bar after every batch\n        if len(metadata_list) == 0:\n            progress_bar.update(batch_size)\n    # Write any remaining metadata to the CSV file\n    if len(metadata_list) > 0:\n        with open(output_path, 'a', newline='') as file:\n            writer = csv.writer(file)\n            for metadata in metadata_list:\n                writer.writerow(metadata)\n        metadata_list = []\n    # Close the progress bar\n    progress_bar.close()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T15:12:45.863473Z","iopub.execute_input":"2023-02-26T15:12:45.863944Z","iopub.status.idle":"2023-02-26T15:59:08.47781Z","shell.execute_reply.started":"2023-02-26T15:12:45.863901Z","shell.execute_reply":"2023-02-26T15:59:08.473805Z"},"trusted":true},"execution_count":null,"outputs":[]}]}