{"cells":[{"metadata":{},"cell_type":"markdown","source":"# VinBigData-starter Ver 1.0\n\nTo start build the predict model, I gonna try to figure out what the dataset look like. This notebook is samplely built to see the picture and the additional information of the dataset."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport pydicom # dicom file process\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport matplotlib.pyplot as plt # plt modules\nimport matplotlib.patches as patches # plt rectangle modules\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Read the Data\n\nThere are two file folders in the dataset, and I just gonna to check train folder, also there are two csv files, one named trian.csv, another named sample_submission.csv. We gonna check these files, too.\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"filePath = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/\"\ncsvPath = \"train.csv\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv = pd.read_csv(filePath + csvPath)\nlen(train_csv)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"filenames = train_csv['image_id']\nfilenames","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_counts = dict()\nfor file in filenames:\n    if file in train_counts:\n        train_counts[file] += 1\n    else:\n        train_counts[file] = 1\ntrain_counts","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# chose one row of train csv files to show\ndef showTrainDicom(train_row, train_csv):\n    ''' visualize the dicom and box labeled of training set\n        trainrow - one specific row of train csv files, e.g 2\n    '''\n    # get dicom name from train_csv\n    images = train_csv['image_id']\n    dicom_name = images[train_row]\n    print('read dicom file: ', dicom_name)\n    # read dicom file\n    traindcm = pydicom.read_file(filePath + \"train/\" + dicom_name + \".dicom\")\n    traindcm\n    # extract information from dicom file\n    info = dict()\n    info['sex'] = traindcm.PatientSex\n#     info['age'] = traindcm.PatientAge\n    info['rows'] = traindcm.Rows\n    info['Columns'] = traindcm.Columns\n    info['pixeldata'] = traindcm.pixel_array\n    # plot the dicom figure\n    plt.figure(figsize=(20, 20))\n    plt.imshow(info['pixeldata'], cmap = plt.cm.gray)\n    # get the labeled data\n    for i in range(len(train_csv)):\n        if train_csv['image_id'][i] == dicom_name:\n            if train_csv['class_id'][i] != 14:\n                info['class_id'] = train_csv['class_id'][i]\n                info['rad_id'] = train_csv['rad_id'][i]\n                info['x_min'] = train_csv['x_min'][i]\n                info['y_min'] = int(train_csv['y_min'][i])\n                info['x_max'] = int(train_csv['x_max'][i])\n                info['y_max'] = int(train_csv['y_max'][i])\n                currentAxis = plt.gca() # get current axis\n                print(info['class_id'], info['rad_id'], info['x_min'], info['x_min'], info['x_max'], info['y_max'])\n                rect = patches.Rectangle((info['x_min'], info['x_min']), (info['x_max']-info['x_min']), (info['y_max']-info['y_min']),linewidth=1,edgecolor='r',facecolor='none')\n                currentAxis.add_patch(rect)\n\n    plt.show()\n    return info","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"info = showTrainDicom(3, train_csv)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"traindcm = pydicom.read_file(filePath + \"train/\" + train_csv['image_id'][19] + \".dicom\")\ntraindcm","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}