{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"---\nDownsize https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-jpg\n\nto 608px and normalize+CLIHE images","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pydicom\nimport os\nimport cv2\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom PIL import Image\nfrom skimage import exposure\nfrom tqdm.auto import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-26T13:47:58.280903Z","iopub.execute_input":"2021-08-26T13:47:58.281293Z","iopub.status.idle":"2021-08-26T13:47:58.286512Z","shell.execute_reply.started":"2021-08-26T13:47:58.281259Z","shell.execute_reply":"2021-08-26T13:47:58.285521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_xray(path, voi_lut = True, fix_monochrome = True):\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    \n    data = data - np.min(data)\n\n    # added\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data","metadata":{"execution":{"iopub.status.busy":"2021-08-26T13:48:00.02775Z","iopub.execute_input":"2021-08-26T13:48:00.028353Z","iopub.status.idle":"2021-08-26T13:48:00.036182Z","shell.execute_reply.started":"2021-08-26T13:48:00.028317Z","shell.execute_reply":"2021-08-26T13:48:00.034781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize(img, size, padColor=0):\n\n    h, w = img.shape[:2]\n    sh, sw = size\n\n    # interpolation method\n    if h > sh or w > sw: # shrinking image\n        interp = cv2.INTER_AREA\n    else: # stretching image\n        interp = cv2.INTER_CUBIC\n\n    # aspect ratio of image\n    aspect = w/h  # if on Python 2, you might need to cast as a float: float(w)/h\n\n    # compute scaling and pad sizing\n    if aspect > 1: # horizontal image\n        new_w = sw\n        new_h = np.round(new_w/aspect).astype(int)\n        pad_vert = (sh-new_h)/2\n        pad_top, pad_bot = np.floor(pad_vert).astype(int), np.ceil(pad_vert).astype(int)\n        pad_left, pad_right = 0, 0\n    elif aspect < 1: # vertical image\n        new_h = sh\n        new_w = np.round(new_h*aspect).astype(int)\n        pad_horz = (sw-new_w)/2\n        pad_left, pad_right = np.floor(pad_horz).astype(int), np.ceil(pad_horz).astype(int)\n        pad_top, pad_bot = 0, 0\n    else: # square image\n        new_h, new_w = sh, sw\n        pad_left, pad_right, pad_top, pad_bot = 0, 0, 0, 0\n\n    # set pad color\n    if len(img.shape) is 3 and not isinstance(padColor, (list, tuple, np.ndarray)): # color image but only one color provided\n        padColor = [padColor]*3\n\n    # scale and pad\n    scaled_img = cv2.resize(img, (new_w, new_h), interpolation=interp)\n    # keep aspect ratio (no padding)\n    # scaled_img = cv2.copyMakeBorder(scaled_img, pad_top, pad_bot, pad_left, pad_right, borderType=cv2.BORDER_CONSTANT, value=padColor)\n\n    return scaled_img","metadata":{"execution":{"iopub.status.busy":"2021-08-26T13:48:01.774498Z","iopub.execute_input":"2021-08-26T13:48:01.774932Z","iopub.status.idle":"2021-08-26T13:48:01.786959Z","shell.execute_reply.started":"2021-08-26T13:48:01.774898Z","shell.execute_reply":"2021-08-26T13:48:01.785773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test 1 img\nimage_id = []\norig_height = []\norig_width = []\nre_height = []\nre_width = []\n\nfor split in ['train']:\n    load_dir = f'../input/vinbigdata-chest-xray-abnormalities-detection/{split}/'\n    save_dir = f'/kaggle/working/{split}/'\n    \n    os.makedirs(save_dir, exist_ok=True)\n\n    for file in tqdm(os.listdir(load_dir)):\n        xray = read_xray(load_dir + file)\n        im = resize(xray, (608,608))  # yolov4 default 608\n        im = exposure.equalize_hist(im) # histogram normalization\n        im = exposure.equalize_adapthist(im/np.max(im)) #clahe\n        cv2.imwrite(save_dir + file.replace('dicom', 'jpg'), im*255)\n        \n        # shape[0] = height, 1 = width\n        if split == 'train':\n            image_id.append(file.replace('.dicom', ''))\n            re_height.append(im.shape[0])\n            re_width.append(im.shape[1])\n            orig_height.append(xray.shape[0])\n            orig_width.append(xray.shape[1])\n            \n            break\n    break","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/working/train/","metadata":{"execution":{"iopub.status.busy":"2021-08-26T11:34:46.757312Z","iopub.execute_input":"2021-08-26T11:34:46.757754Z","iopub.status.idle":"2021-08-26T11:34:47.512648Z","shell.execute_reply.started":"2021-08-26T11:34:46.757717Z","shell.execute_reply":"2021-08-26T11:34:47.511479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_resized = pd.DataFrame.from_dict({\n    'image_id': image_id, \n    're_height': re_height, \n    're_width': re_width,\n    'orig_height': orig_height,\n    'orig_width': orig_width\n})\ndf_resized","metadata":{"execution":{"iopub.status.busy":"2021-08-26T11:35:27.967896Z","iopub.execute_input":"2021-08-26T11:35:27.968293Z","iopub.status.idle":"2021-08-26T11:35:27.98948Z","shell.execute_reply.started":"2021-08-26T11:35:27.968252Z","shell.execute_reply":"2021-08-26T11:35:27.988315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# resize\nimage_id = []\norig_height = []\norig_width = []\nre_height = []\nre_width = []\n\nfor split in ['train', 'test']:\n    load_dir = f'../input/vinbigdata-chest-xray-abnormalities-detection/{split}/'\n    save_dir = f'/kaggle/tmp/{split}/'\n#     save_dir = f'/kaggle/working/{split}/'\n\n\n    os.makedirs(save_dir, exist_ok=True)\n\n    for file in tqdm(os.listdir(load_dir)):\n        xray = read_xray(load_dir + file)\n        im = resize(xray, (608,608))  # yolov4 default 608\n        im = exposure.equalize_hist(im) # histogram normalization\n        im = exposure.equalize_adapthist(im/np.max(im)) #clahe\n        cv2.imwrite(save_dir + file.replace('dicom', 'jpg'), im*255)\n        \n        # shape[0] = height, 1 = width\n        if split == 'train':\n            image_id.append(file.replace('.dicom', ''))\n            re_height.append(im.shape[0])\n            re_width.append(im.shape[1])\n            orig_height.append(xray.shape[0])\n            orig_width.append(xray.shape[1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_resized = pd.DataFrame.from_dict({\n    'image_id': image_id, \n    're_height': re_height, \n    're_width': re_width,\n    'orig_height': orig_height,\n    'orig_width': orig_width\n})\n","metadata":{"execution":{"iopub.status.busy":"2021-08-26T11:48:07.656296Z","iopub.execute_input":"2021-08-26T11:48:07.656926Z","iopub.status.idle":"2021-08-26T11:48:07.667152Z","shell.execute_reply.started":"2021-08-26T11:48:07.65688Z","shell.execute_reply":"2021-08-26T11:48:07.665888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\nclean up","metadata":{}},{"cell_type":"code","source":"%cd /kaggle/tmp/\n!ls","metadata":{"execution":{"iopub.status.busy":"2021-08-26T11:59:35.679224Z","iopub.execute_input":"2021-08-26T11:59:35.679686Z","iopub.status.idle":"2021-08-26T11:59:36.421508Z","shell.execute_reply.started":"2021-08-26T11:59:35.679628Z","shell.execute_reply":"2021-08-26T11:59:36.420271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle/tmp/train/\ndir = f'/kaggle/tmp/train/'\n\nfor file in os.listdir(dir):\n    im_id = file[:-4]\n    print(file)\n    break\nim_id","metadata":{"execution":{"iopub.status.busy":"2021-08-26T11:59:38.247117Z","iopub.execute_input":"2021-08-26T11:59:38.247503Z","iopub.status.idle":"2021-08-26T11:59:38.270722Z","shell.execute_reply.started":"2021-08-26T11:59:38.247463Z","shell.execute_reply":"2021-08-26T11:59:38.269975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shape[0] = height, 1 = width\n# resize\nimage_id = []\n# orig_height = []\n# orig_width = []\nre_height = []\nre_width = []\n\ndir = f'/kaggle/tmp/train/'\nfor file in tqdm(os.listdir(dir)):\n    im_id = file[:-4]\n    im =  cv2.imread(file)\n\n    image_id.append(im_id)\n    re_height.append(im.shape[0])\n    re_width.append(im.shape[1])","metadata":{"execution":{"iopub.status.busy":"2021-08-26T11:59:46.771982Z","iopub.execute_input":"2021-08-26T11:59:46.772529Z","iopub.status.idle":"2021-08-26T12:01:04.113985Z","shell.execute_reply.started":"2021-08-26T11:59:46.772478Z","shell.execute_reply":"2021-08-26T12:01:04.112959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_resized = pd.DataFrame.from_dict({\n    'image_id': image_id, \n    're_height': re_height, \n    're_width': re_width\n})","metadata":{"execution":{"iopub.status.busy":"2021-08-26T12:01:26.519085Z","iopub.execute_input":"2021-08-26T12:01:26.51945Z","iopub.status.idle":"2021-08-26T12:01:26.544335Z","shell.execute_reply.started":"2021-08-26T12:01:26.519418Z","shell.execute_reply":"2021-08-26T12:01:26.542772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\nCreate yolov4 txt files\n\nhttps://www.kaggle.com/jackpodkim/vbd-convert-labels-to-yolo-yolov4/edit","metadata":{}},{"cell_type":"code","source":"%cd /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2021-08-26T22:57:39.001772Z","iopub.execute_input":"2021-08-26T22:57:39.00228Z","iopub.status.idle":"2021-08-26T22:57:39.013506Z","shell.execute_reply.started":"2021-08-26T22:57:39.002185Z","shell.execute_reply":"2021-08-26T22:57:39.012229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport pydicom\nimport glob\n\ndf = pd.read_csv(\"../input/vinbigdata-chest-xray-abnormalities-detection/train.csv\")\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-26T22:57:41.967049Z","iopub.execute_input":"2021-08-26T22:57:41.96733Z","iopub.status.idle":"2021-08-26T22:57:42.478183Z","shell.execute_reply.started":"2021-08-26T22:57:41.967306Z","shell.execute_reply":"2021-08-26T22:57:42.476618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicom_metadata = [pydicom.filereader.dcmread(f\"../input/vinbigdata-chest-xray-abnormalities-detection/train/{image_id}.dicom\", stop_before_pixels=True) for image_id in df['image_id']]","metadata":{"execution":{"iopub.status.busy":"2021-08-26T22:57:56.592691Z","iopub.execute_input":"2021-08-26T22:57:56.592977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['orig_width'] = [i.Columns for i in dicom_metadata]\ndf['orig_height'] = [i.Rows for i in dicom_metadata]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df[df.class_id!=14].reset_index(drop = True)\n\nprint(\"We have {} unique images with boxes.\".format(len(df.image_id.unique())))\nunique_img_ids = df.image_id.unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.merge(df_resized, how='left', on='image_id')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head().T","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # resized reindex\n# df['x'] = df.apply(lambda row: row.x_min*(row.re_width/row.orig_width), axis =1)\n# df['y'] = df.apply(lambda row: row.y_min*(row.re_height/row.orig_height), axis =1)\n\n# df['x_re_max'] = df.apply(lambda row: row.x_max*(row.re_width/row.orig_width), axis =1)\n# df['y_re_max'] = df.apply(lambda row: row.y_max*(row.re_height/row.orig_height), axis =1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # resized reindex\n# df['x_re_min'] = df.apply(lambda row: row.x_min*(row.re_width_x/row.orig_width), axis =1)\n# df['y_re_min'] = df.apply(lambda row: row.y_min*(row.re_height_x/row.orig_height), axis =1)\n\n# df['x_re_max'] = df.apply(lambda row: row.x_max*(row.re_width_x/row.orig_width), axis =1)\n# df['y_re_max'] = df.apply(lambda row: row.y_max*(row.re_height_x/row.orig_height), axis =1)","metadata":{"execution":{"iopub.status.busy":"2021-08-26T12:51:50.882555Z","iopub.execute_input":"2021-08-26T12:51:50.883026Z","iopub.status.idle":"2021-08-26T12:51:56.269501Z","shell.execute_reply.started":"2021-08-26T12:51:50.882985Z","shell.execute_reply":"2021-08-26T12:51:56.268445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# yolov4 format\ndf['x_mid'] = df.apply(lambda row: (row.x_min+row.x_max)/2, axis =1)\ndf['y_mid'] = df.apply(lambda row: (row.y_re_max+row.y_re_min)/2, axis =1)\n\n# df['w'] = df.apply(lambda row: (row.x_re_max-row.x_re_min), axis =1)\n# df['h'] = df.apply(lambda row: (row.y_re_max-row.y_re_min), axis =1)\n\n# df['area'] = df['w']*df['h']\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-26T12:52:07.497565Z","iopub.execute_input":"2021-08-26T12:52:07.49802Z","iopub.status.idle":"2021-08-26T12:52:11.515576Z","shell.execute_reply.started":"2021-08-26T12:52:07.497982Z","shell.execute_reply":"2021-08-26T12:52:11.513984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['yolo_box'] = df[['x_mid', 'y_mid', 'w', 'h']].values.tolist()\n\nprint(\"We have {} unique images with boxes.\".format(len(df.image_id.unique())))\nunique_img_ids = df.image_id.unique()","metadata":{"execution":{"iopub.status.busy":"2021-08-26T12:52:29.080182Z","iopub.execute_input":"2021-08-26T12:52:29.080547Z","iopub.status.idle":"2021-08-26T12:52:29.136698Z","shell.execute_reply.started":"2021-08-26T12:52:29.080514Z","shell.execute_reply":"2021-08-26T12:52:29.135497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle/tmp/","metadata":{"execution":{"iopub.status.busy":"2021-08-26T12:52:32.204526Z","iopub.execute_input":"2021-08-26T12:52:32.204946Z","iopub.status.idle":"2021-08-26T12:52:32.215529Z","shell.execute_reply.started":"2021-08-26T12:52:32.20491Z","shell.execute_reply":"2021-08-26T12:52:32.213639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folder_location = \"/kaggle/tmp/train/\"\n\nfor img_id in tqdm(unique_img_ids): # loop through all unique image ids. Remove the slice to do all images\n    filt_df = df.query(\"image_id == @img_id\") # filter the df to a specific id\n    #all_boxes = filt_df.yolo_box.values\n    file_name = \"{}/{}.txt\".format(folder_location,img_id) # specify the name of the folder and get a file name\n\n    with open(file_name, 'w+') as file: # append lines to file\n        for i in filt_df.iterrows():\n            s = f\"{i[1].class_id} %s %s %s %s \\n\" # The first number is the class name\n            new_line = (s % tuple(i[1].yolo_box))\n            file.write(new_line)","metadata":{"execution":{"iopub.status.busy":"2021-08-26T12:52:36.120926Z","iopub.execute_input":"2021-08-26T12:52:36.121322Z","iopub.status.idle":"2021-08-26T12:53:05.588538Z","shell.execute_reply.started":"2021-08-26T12:52:36.121287Z","shell.execute_reply":"2021-08-26T12:53:05.587197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/tmp/train/","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create labels for training images that do not have bounding boxes\n# If you wish to train on only images with a finding, remove this code cell\nall_imgs = glob.glob(\"../input/vinbigdata-chest-xray-abnormalities-detection/train/*.dicom\")\nall_imgs = [i.split(\"/\")[-1].replace(\".dicom\", \"\") for i in all_imgs]\npositive_imgs = df.image_id.unique()\n\nnegative_images = set(all_imgs) - set(positive_imgs)\nprint('All images:', len(all_imgs), 'Positive images:', len(positive_imgs))\n\nfor i in tqdm(list(negative_images)):\n    file_name = \"{}/{}.txt\".format(folder_location, i)\n    #print(file_name)\n    with open(file_name, 'w') as fp:\n        pass","metadata":{"execution":{"iopub.status.busy":"2021-08-26T12:55:10.124957Z","iopub.execute_input":"2021-08-26T12:55:10.12535Z","iopub.status.idle":"2021-08-26T12:55:10.85069Z","shell.execute_reply.started":"2021-08-26T12:55:10.125315Z","shell.execute_reply":"2021-08-26T12:55:10.849437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n\n# zip to make files easier to download\n\n!zip -r yolo_labels.zip /kaggle/tmp/","metadata":{"execution":{"iopub.status.busy":"2021-08-26T12:55:16.898617Z","iopub.execute_input":"2021-08-26T12:55:16.898992Z","iopub.status.idle":"2021-08-26T12:57:09.670506Z","shell.execute_reply.started":"2021-08-26T12:55:16.898961Z","shell.execute_reply":"2021-08-26T12:57:09.668996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mv /kaggle/tmp/yolo_labels.zip /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2021-08-26T12:57:09.672686Z","iopub.execute_input":"2021-08-26T12:57:09.673021Z","iopub.status.idle":"2021-08-26T12:57:13.255559Z","shell.execute_reply.started":"2021-08-26T12:57:09.672985Z","shell.execute_reply":"2021-08-26T12:57:13.253938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\nclean up","metadata":{}},{"cell_type":"code","source":"# resize\nimage_id = []\norig_height = []\norig_width = []\nre_height = []\nre_width = []\n\n#test\nload_dir = f'../input/vinbigdata-chest-xray-abnormalities-detection/test/'\nsave_dir = f'/kaggle/working/test/'\n#     save_dir = f'/kaggle/working/{split}/'\n\n\nos.makedirs(save_dir, exist_ok=True)\n\nfor file in tqdm(os.listdir(load_dir)):\n    xray = read_xray(load_dir + file)\n    im = resize(xray, (608,608))  # yolov4 default 608\n    im = exposure.equalize_hist(im) # histogram normalization\n    im = exposure.equalize_adapthist(im/np.max(im)) #clahe\n    cv2.imwrite(save_dir + file.replace('dicom', 'jpg'), im*255)","metadata":{"execution":{"iopub.status.busy":"2021-08-26T13:48:16.773966Z","iopub.execute_input":"2021-08-26T13:48:16.774372Z","iopub.status.idle":"2021-08-26T14:36:11.716511Z","shell.execute_reply.started":"2021-08-26T13:48:16.774338Z","shell.execute_reply":"2021-08-26T14:36:11.714297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n\n# zip to make files easier to download\n\n!zip -r yolo_test.zip /kaggle/working/test","metadata":{"execution":{"iopub.status.busy":"2021-08-26T13:32:53.366208Z","iopub.execute_input":"2021-08-26T13:32:53.36658Z","iopub.status.idle":"2021-08-26T13:32:54.152417Z","shell.execute_reply.started":"2021-08-26T13:32:53.366548Z","shell.execute_reply":"2021-08-26T13:32:54.150757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2021-08-26T14:51:42.179491Z","iopub.execute_input":"2021-08-26T14:51:42.180389Z","iopub.status.idle":"2021-08-26T14:51:43.007237Z","shell.execute_reply.started":"2021-08-26T14:51:42.180258Z","shell.execute_reply":"2021-08-26T14:51:43.005811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}