{"cells":[{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"### Imports\nimport os\nimport scipy\nimport pandas as pd\n# import pydicom\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport vtk\nfrom vtk.util import numpy_support\nimport cv2\nimport SimpleITK as sitk\n!pip install git+https://github.com/JoHof/lungmask\nfrom lungmask import mask    \nfrom skimage.morphology import convex_hull_image\nfrom skimage.transform import resize\nimport h5py\n#!conda install -c conda-forge gdcm -y","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# Preprocess\n# load basepath\nNUM_DS = 7279\nbasepath = \"../input/rsna-str-pulmonary-embolism-detection/\"\n# load CSV labels\ntrain = pd.read_csv(basepath + \"train.csv\")\ntest = pd.read_csv(basepath + \"test.csv\")\n# create new column\ntrain[\"dcm_path\"] = basepath + \"train/\" + train.StudyInstanceUID + \"/\" + train.SeriesInstanceUID;","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"reader = vtk.vtkDICOMImageReader()\ndef load_scans_VTK(PathDicom):\n    reader.SetDirectoryName(PathDicom)\n    reader.Update()\n\n    # Load dimensions using `GetDataExtent`\n    _extent = reader.GetDataExtent()\n    ConstPixelDims = [_extent[1]-_extent[0]+1, _extent[3]-_extent[2]+1, _extent[5]-_extent[4]+1]\n\n    # Load spacing values\n    ConstPixelSpacing = reader.GetPixelSpacing()\n\n    # Get the 'vtkImageData' object from the reader\n    imageData = reader.GetOutput()\n    # Get the 'vtkPointData' object from the 'vtkImageData' object\n    pointData = imageData.GetPointData()\n    # Ensure that only one array exists within the 'vtkPointData' object\n    assert (pointData.GetNumberOfArrays()==1)\n    # Get the `vtkArray` (or whatever derived type) which is needed for the `numpy_support.vtk_to_numpy` function\n    arrayData = pointData.GetArray(0)\n\n    # Convert the `vtkArray` to a NumPy array\n    ArrayDicom = numpy_support.vtk_to_numpy(arrayData)\n    # Reshape the NumPy array to 3D using 'ConstPixelDims' as a 'shape'\n    ArrayDicom = ArrayDicom.reshape((512, 512,-1), order='F')\n    ArrayDicom[~(ArrayDicom==0).all((2,1))]\n    return ArrayDicom\n# Set outside circle to air.\ndef set_outside_scanner_to_air(raw_pixelarrays):\n    # in OSIC we find outside-scanner-regions with raw-values of -2000. \n    # Let's threshold between air (0) and this default (-2000) using -1000\n    raw_pixelarrays[raw_pixelarrays <= -1000] = 0\n    return raw_pixelarrays\ndef flood_fill_hull(image):    \n    points = np.transpose(np.where(image))\n    hull = scipy.spatial.ConvexHull(points)\n    deln = scipy.spatial.Delaunay(points[hull.vertices]) \n    idx = np.stack(np.indices(image.shape), axis = -1)\n    out_idx = np.nonzero(deln.find_simplex(idx) + 1)\n    out_img = np.zeros(image.shape)\n    out_img[out_idx] = 1\n    return out_img, hull","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# initial model is exam level, so get one row for every exam\nndTrain = train.drop_duplicates(subset='StudyInstanceUID', keep=\"first\")\nndTrain['positive_exam_for_pe'] = ~ndTrain['negative_exam_for_pe'].astype('bool') & ~ndTrain['indeterminate'].astype('bool')\nnnTrainLabels = [ndTrain['negative_exam_for_pe'], ndTrain['positive_exam_for_pe'].astype('int64'), ndTrain['indeterminate']]\nnnTrainLabels = pd.concat(nnTrainLabels, axis=1)\nnnTrainLabels = nnTrainLabels.to_numpy()\nprint(nnTrainLabels.shape)\n# save to HDF5 later on","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"!pip install pydrive\nimport os\nos.makedirs('../outputs')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import json\ndictionary = {\"web\":{\"client_id\":\"700759027193-j03r5nq1q7a5q7q2vo7seo06tir7pfg8.apps.googleusercontent.com\",\"project_id\":\"pure-polymer-292623\",\"auth_uri\":\"https://accounts.google.com/o/oauth2/auth\",\"token_uri\":\"https://oauth2.googleapis.com/token\",\"auth_provider_x509_cert_url\":\"https://www.googleapis.com/oauth2/v1/certs\",\"client_secret\":\"JaBWW9uZ_SL5C5qxSetfkpFa\",\"redirect_uris\":[\"http://localhost:8080/\"],\"javascript_origins\":[\"http://localhost:8080\"]}}\n# Serializing json  \njson_object = json.dumps(dictionary, indent = 4) \n  \n# Writing to sample.json \nwith open(\"/kaggle/working/client_secrets.json\", \"w\") as outfile: \n    outfile.write(json_object) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Program to show various ways to read and \n# write data in a file. \nfile1 = open(\"mycreds.txt\",\"w\") \nL = ['{\"access_token\": \"ya29.a0AfH6SMA0iZ4r4NyQW6BuBF0rlnBrtPj9WQwqsfDB4P_BXKDEjZTnfAHCazzne-saukQlWe-C68KggTaXg5xwtvAZfDdUe7O44WItm3pHSejIAk1D3Si3UQt4E_9KYdxKoAok7c-c8Q49AwwfgXzFRJIbCJ3URMAGzWs\", \"client_id\": \"700759027193-j03r5nq1q7a5q7q2vo7seo06tir7pfg8.apps.googleusercontent.com\", \"client_secret\": \"JaBWW9uZ_SL5C5qxSetfkpFa\", \"refresh_token\": null, \"token_expiry\": \"2020-10-15T04:45:45Z\", \"token_uri\": \"https://oauth2.googleapis.com/token\", \"user_agent\": null, \"revoke_uri\": \"https://oauth2.googleapis.com/revoke\", \"id_token\": null, \"id_token_jwt\": null, \"token_response\": {\"access_token\": \"ya29.a0AfH6SMA0iZ4r4NyQW6BuBF0rlnBrtPj9WQwqsfDB4P_BXKDEjZTnfAHCazzne-saukQlWe-C68KggTaXg5xwtvAZfDdUe7O44WItm3pHSejIAk1D3Si3UQt4E_9KYdxKoAok7c-c8Q49AwwfgXzFRJIbCJ3URMAGzWs\", \"expires_in\": 3599, \"scope\": \"https://www.googleapis.com/auth/drive\", \"token_type\": \"Bearer\"}, \"scopes\": [\"https://www.googleapis.com/auth/drive\"], \"token_info_uri\": \"https://oauth2.googleapis.com/tokeninfo\", \"invalid\": false, \"_class\": \"OAuth2Credentials\", \"_module\": \"oauth2client.client\"}']\nfile1.writelines(L) \nfile1.close() #to change file access modes ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"# create new h5\nf = h5py.File('../outputs/traindata-part1.h5', 'w')\ndset_images = f.create_dataset(\"images\", (1000, 64, 224, 224), chunks=(1,64, 224,224))\ndset_labels = f.create_dataset(\"labels\", (1000, 3))\n# 1 - 1000\ndset_labels = nnTrainLabels\n# save all the images after removing lung and bwconvhull 7280 NUM_DS+1\nfor i in range(0, 1000):\n    tempLoaded = load_scans_VTK(ndTrain.dcm_path.values[i])\n    tempLoaded = set_outside_scanner_to_air(tempLoaded)\n    segmentation = mask.apply(sitk.GetImageFromArray(np.transpose(tempLoaded,(2,1, 0))))\n    # orient\n    tempLoaded = np.rot90(tempLoaded)\n    segmentation = np.flip(segmentation,1)\n    segmentation = np.transpose(segmentation,(1,2,0))\n    # Remove 2s\n    segmentation[segmentation >= 1] = 1\n    # convex hull\n    segmentation,h = flood_fill_hull(segmentation)\n    # mask * data\n    tempLHData = segmentation * tempLoaded\n    lungIndices = np.where(tempLHData.any(axis=(0,1)))[0]\n    tempLHData = tempLHData[:,:,lungIndices]\n    #print(tempLHData.shape)\n    tempLHData = resize(tempLHData, (224, 224, 64), anti_aliasing=False)\n    #print(tempLHData.shape)\n    tempLHData = np.transpose(tempLHData,(2,0,1))\n    #print(tempLHData.shape)\n    #tempLHData = np.expand_dims(tempLHData, 3)\n    #print(tempLHData.shape)\n    #tempLHData = np.concatenate([tempLHData, tempLHData, tempLHData], axis=3)\n    #print(tempLHData.shape)\n    dset_images[i,:,:,:] = tempLHData\n    if i%50 == 0:\n        print('===================== done with:', i, '=====================')\nf.close()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from pydrive.auth import GoogleAuth\n\ngauth = GoogleAuth()\ngauth.LoadCredentialsFile(\"mycreds.txt\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from pydrive.drive import GoogleDrive\ndrive = GoogleDrive(gauth)\n\ncool_image = drive.CreateFile({'title': 'traindata-part1.h5'})\ncool_image.SetContentFile('../outputs/traindata-part1.h5') # load local file data into the File instance\ncool_image.Upload() # creates a file in your drive with the name: my-awesome-file.txt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}