{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Simple function to get DICOM Metadata"},{"metadata":{},"cell_type":"markdown","source":"ElementKeys is an an alphabetical list of element keywords. \n\nI take some infomations about Sex, Age and Weight and convert it to DataFrame. \n\nHope this comes in handy for you!"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np \nimport pandas as pd \n!pip install dicom\nimport pydicom as dcm\nimport pandas as pd\nimport glob\nimport os\nfrom tqdm import tqdm\nimport cv2\nfrom PIL import Image","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Check out https://pydicom.github.io/pydicom/stable/tutorials/dataset_basics.html for more informations."},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"LIMIT = 10 #since the dataset is large, it can be set to small number for testing purpose or 0 to get all the data\n\ndef extract_DICOM_attributes(folder_path):\n    text_file = open(\"dicom_metadata.txt\", \"w\")\n    cnt = 0\n    images = list(os.listdir(folder_path))\n    df = pd.DataFrame()\n    \n    for image in images:\n        cnt += 1\n        if cnt == LIMIT: \n            break\n            \n        image_name = image.split(\".\")[0]\n        dicom_file_path = os.path.join(folder_path,image)\n        dicom_file_dataset = dcm.read_file(dicom_file_path)\n        text_file.write(str(dicom_file_dataset))\n        rows = dicom_file_dataset.Rows\n        columns = dicom_file_dataset.Columns\n        \n        ElementKeys = dicom_file_dataset.dir()\n        print(ElementKeys)\n        \n        PatientSex = dicom_file_dataset.PatientSex if 'PatientSex' in ElementKeys else \"\"\n        PatientWeight = dicom_file_dataset.PatientWeight if 'PatientWeight' in ElementKeys else \"\"\n        PatientAge = dicom_file_dataset.PatientAge if 'PatientAge' in ElementKeys else \"\"\n        PhotometricInterpretation = dicom_file_dataset.PhotometricInterpretation if 'PhotometricInterpretation' in ElementKeys else \"\"\n        Rows = dicom_file_dataset.Rows\n        Columns = dicom_file_dataset.Columns\n#         for k, v in dicom_file_dataset.items():\n#             print(k, v) \n\n        df = df.append(pd.DataFrame({'Image_name': image_name, \n                        'PatientSex': PatientSex,'PatientWeight':PatientWeight, 'PatientAge': PatientAge, 'PhotometricInterpretation': PhotometricInterpretation,\n                        'Rows': Rows,'Columns': Columns}, index = [cnt]))\n    text_file.close()\n    print(df)\n    return df\nextract_DICOM_attributes('../input/vinbigdata-chest-xray-abnormalities-detection/train')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}