{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pydicom\n#ds = pydicom.dcmread('/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/063319de25ce7edb9b1c6b8881290140.dicom')\n#plt.imshow(ds.pixel_array, cmap=plt.cm.bone) \nplt.imshow(result_list[0], cmap=plt.cm.bone) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import vtk\nfrom vtk.util import numpy_support\nimport numpy\n\nPathDicom = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/\"\nreader = vtk.vtkDICOMImageReader()\nreader.SetDirectoryName(PathDicom)\nreader.Update()\n# Load dimensions using `GetDataExtent`\n_extent = reader.GetDataExtent()\nConstPixelDims = [_extent[1]-_extent[0]+1, _extent[3]-_extent[2]+1, _extent[5]-_extent[4]+1]\n\n# Load spacing values\nConstPixelSpacing = reader.GetPixelSpacing()\n# Get the 'vtkImageData' object from the reader\nimageData = reader.GetOutput()\n# Get the 'vtkPointData' object from the 'vtkImageData' object\npointData = imageData.GetPointData()\n# Ensure that only one array exists within the 'vtkPointData' object\n#assert (pointData.GetNumberOfArrays()==1)\n# Get the `vtkArray` (or whatever derived type) which is needed for the `numpy_support.vtk_to_numpy` function\narrayData = pointData.GetArray(1)\n\n# Convert the `vtkArray` to a NumPy array\nArrayDicom = numpy_support.vtk_to_numpy(arrayData)\n# Reshape the NumPy array to 3D using 'ConstPixelDims' as a 'shape'\nArrayDicom = ArrayDicom.reshape(ConstPixelDims, order='F')\nArrayDicom","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nfrom vtk.numpy_interface import dataset_adapter as dsa\n\n\na = dsa.WrapDataObject(imageData)\nprint(a.PointData.keys())\nvt = a.PointData['DICOMImage']\nvt[0]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\n\ndf = pd.read_csv('/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv')\ndf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['class_name'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"foldertrain = '/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/'\n\ndf['folder_train'] = '/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/' + df['image_id'] + '.dicom'\n\ndf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install dicom_parser","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":true,"trusted":true},"cell_type":"code","source":"from dicom_parser import Image\nimage = Image('/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/063319de25ce7edb9b1c6b8881290140.dicom')\n#for k,v in image.header.raw.items():\n    #print(k,v)\n#image.header.raw[list(image.header.raw.keys())[-1]].value\nimage._data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-output":true},"cell_type":"code","source":"from pydicom import dcmread\nfrom dicom_parser import Image\n\ndf2 = df[:10000]\ndf2['folder_train'].count()\ndf2['folder_train_processed'] = df2['folder_train'].map(lambda x : Image(x))\ndf2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from pydicom import dcmread\nimport multiprocess as mp\nfrom pydicom import dcmread\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport numpy as np\n\ndef dicom2array(path):\n    voi_lut=True\n    fix_monochrome=True\n    dicom = pydicom.read_file(path)\n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\nresult_list = []\ndef log_result(result):\n    # This is called whenever foo_pool(i) returns a result.\n    # result_list is modified only by the main process, not the pool workers.\n    result_list.append(result)\n\n\ndf2 = df[:1001]\nli_to_proc = df2['folder_train'].to_list()\nprint(len(li_to_proc))\npool = mp.Pool(processes=5)#,maxtasksperchild=1000)\ndcms_info = [pool.apply_async(dicom2array, args=(dcm_path,), callback = log_result) for dcm_path in li_to_proc]\n\npool.close()\npool.join()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(result_list)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"xx= [[0,1001]]\nk = [ [xx[0][1]*x-(x)+1, xx[0][1]*(x+1)-(x)] for x in range(10) ]\nfor xk in k:\n    print(xk)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from pydicom import dcmread\nimport multiprocess as mp\nfrom pydicom import dcmread\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport numpy as np\nimport os\nimport gc\nimport psutil\nimport pickle\n\ndef dicom2array(path):\n    voi_lut=True\n    fix_monochrome=True\n    dicom = pydicom.read_file(path)\n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\nresult_list = []\ndef log_result(result):\n    # This is called whenever foo_pool(i) returns a result.\n    # result_list is modified only by the main process, not the pool workers.\n    result_list.append(result)\n    \nxx= [[0,500]]\nk = [ [xx[0][1]*x-(x)+1, xx[0][1]*(x+1)-(x)] for x in range(10) ]\nfor xk in k:\n    df2 = df[xk[0]:xk[1]]\n    li_to_proc = df2['folder_train'].to_list()\n    print(len(li_to_proc))\n    pool = mp.Pool(processes=5,maxtasksperchild=10)\n    dcms_info = [pool.apply_async(dicom2array, args=(dcm_path,), callback = log_result) for dcm_path in li_to_proc] \n\n    pool.close()\n    pool.join()\n    pool.terminate()\n    del pool\n    proc = psutil.Process(os.getpid())\n    gc.collect()\n    with open('l1.pkl', 'wb') as f:\n        pickle.dump(result_list, f)\n   \n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf \n\ndef dicom2array(path):\n    voi_lut=True\n    fix_monochrome=True\n    print(path)\n    dicom = pydicom.read_file(path)\n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\nfilenames_dataset = tf.data.Dataset.list_files(df['folder_train'].to_list())\nimage_dataset = filenames_dataset.map(dicom2array)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"xx= [[0,500]]\nk = [ [xx[0][1]*x-(x)+1, xx[0][1]*(x+1)-(x)] for x in range(10) ]\nk","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf\nbatch_size = 32\nimg_height = 224\nimg_width = 224\n\n\n\ntrain_list_ds = tf.data.Dataset.from_tensor_slices((total_x, total))\n\n\ntrain_list_ds = tf.keras.preprocessing.image_dataset_from_directory(\n  './',\n  validation_split=0.3,\n  subset=\"training\",\n  seed=123,\n  image_size=(img_height, img_width),\n  batch_size=batch_size)\n\nval_ds = tf.keras.preprocessing.image_dataset_from_directory(\n  './',\n  validation_split=0.3,\n  subset=\"validation\",\n  seed=123,\n  image_size=(img_height, img_width),\n  batch_size=batch_size)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}