{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np, pandas as pd, pydicom as dcm\nimport matplotlib.pyplot as plt, seaborn as sns\nimport os, glob","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/rsna-str-pulmonary-embolism-detection/train.csv\")\ntest_df = pd.read_csv(\"../input/rsna-str-pulmonary-embolism-detection/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Insights on tabular data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f\"Total no. of studies = {train_df['StudyInstanceUID'].unique().shape[0]}\")\nprint(f\"Total no. of series = {train_df['SeriesInstanceUID'].unique().shape[0]}\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"This shows that total studies = series which implies studies and series are one and the same (uniquely identify the study)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"x = train_df.groupby('StudyInstanceUID').mean()\nx.sum()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"From this we can confirm that 'pe_present_on_image' label is image specific and rest of the  labels are study specific","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# for i in ['qa_motion', 'qa_contrast', 'true_filling_defect_not_pe', 'flow_artifact']\nprint(f\"Total indeterminate using qa_motion and qa_contrast = {train_df[(train_df['qa_motion']== 1.0) | (train_df['qa_contrast']== 1.0)].shape[0]}\")\nprint(f\"Total indeterminate directly = {train_df[(train_df['indeterminate']== 1.0)].shape[0]}\")\n\nprint(f\"Total indeterminate and has PE = {train_df[(train_df['indeterminate']== 1.0) & (train_df['pe_present_on_image']== 1.0)].shape[0]}\")\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"From this we can say that just predicting the indeterminate label would suffice and all the studies with indeterminate flag = 1 have no PE images","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f\"Total Flow artifact and has PE = {train_df[(train_df['indeterminate']== 1.0) & (train_df['flow_artifact']== 1.0)].shape[0]}\")\nprint(f\"Total True filling defect not PE and has PE = {train_df[(train_df['indeterminate']== 1.0) & (train_df['true_filling_defect_not_pe']== 1.0)].shape[0]}\")\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"> They are QA Contrast, QA Motion, True filling defect not PE, and Flow artifact, and are not scored, but are meant to be used as helpers. Also note that Acute PE is not an explicit label, but is implied by the lack of Chronic PE or Acute and Chronic PE.\n\nwe may not need the QA Contrast, QA Motion, flow artifact label but True filling defect not PE predictor may help us to find non PE images/studies directly","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# Insights on Images data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import pydicom as dicom\n# Load the scans in given folder path\ndef load_scan(path):\n    slices = [dicom.read_file(path + '/' + s) for s in os.listdir(path)]\n    slices.sort(key = lambda x: float(x.ImagePositionPatient[2]))\n    try:\n        slice_thickness = np.abs(slices[0].ImagePositionPatient[2] - slices[1].ImagePositionPatient[2])\n    except:\n        slice_thickness = np.abs(slices[0].SliceLocation - slices[1].SliceLocation)\n        \n    for s in slices:\n        s.SliceThickness = slice_thickness\n        \n    return slices\n\ndef get_pixels_hu(slices):\n    image = np.stack([s.pixel_array for s in slices])\n    # Convert to int16 (from sometimes int16), \n    # should be possible as values should always be low enough (<32k)\n    image = image.astype(np.int16)\n\n    # Set outside-of-scan pixels to 0\n    # The intercept is usually -1024, so air is approximately 0\n    image[image == -2000] = 0\n    \n    # Convert to Hounsfield units (HU)\n    for slice_number in range(len(slices)):\n        \n        intercept = slices[slice_number].RescaleIntercept\n        slope = slices[slice_number].RescaleSlope\n        \n        if slope != 1:\n            image[slice_number] = slope * image[slice_number].astype(np.float64)\n            image[slice_number] = image[slice_number].astype(np.int16)\n            \n        image[slice_number] += np.int16(intercept)\n    \n    return np.array(image, dtype=np.int16)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"first_patient = load_scan('../input/rsna-str-pulmonary-embolism-detection/train/0003b3d648eb/d2b2960c2bbf')\nfirst_patient_pixels = get_pixels_hu(first_patient)\n\nplt.imshow(first_patient_pixels[80], cmap=plt.cm.gray)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, plots = plt.subplots(8, 10, sharex='col', sharey='row', figsize=(20, 16))\nfor i in range(80):\n    plots[i // 10, i % 10].axis('off')\n    plots[i // 10, i % 10].imshow(first_patient_pixels[i], cmap=plt.cm.bone) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"**From this we can say that each study => each patient**","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"More insights on the way...","execution_count":null}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}