{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import matplotlib.patches as patches\n\nimport numpy as np \nimport pandas as pd\nfrom glob import glob\n\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport gc\nimport os\nfrom PIL import Image\nimport cv2\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2021-06-13T10:17:32.895362Z","iopub.execute_input":"2021-06-13T10:17:32.895901Z","iopub.status.idle":"2021-06-13T10:17:33.391189Z","shell.execute_reply.started":"2021-06-13T10:17:32.895845Z","shell.execute_reply":"2021-06-13T10:17:33.390392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"KAGGLE=False\nPATH=\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/\" if KAGGLE else \"./\"","metadata":{"execution":{"iopub.status.busy":"2021-06-13T10:17:33.392641Z","iopub.execute_input":"2021-06-13T10:17:33.393096Z","iopub.status.idle":"2021-06-13T10:17:33.396495Z","shell.execute_reply.started":"2021-06-13T10:17:33.393049Z","shell.execute_reply":"2021-06-13T10:17:33.395681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.read_csv(PATH+'train.csv')\nsub=pd.read_csv(PATH+'sample_submission.csv')","metadata":{"scrolled":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train['class_id'] != 14].isnull().sum()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.class_name.unique()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> The dataset comprises 18,000 postero-anterior (PA) CXR scans in DICOM format, which were de-identified to protect patient privacy. All images were labeled by a panel of experienced radiologists for the presence of 14 critical radiographic findings as listed below:\n\n0 - Aortic enlargement\n\n1 - Atelectasis\n\n2 - Calcification\n\n3 - Cardiomegaly\n\n4 - Consolidation\n\n5 - ILD\n\n6 - Infiltration\n\n7 - Lung Opacity\n\n8 - Nodule/Mass\n\n9 - Other lesion\n\n10 - Pleural effusion\n\n11 - Pleural thickening\n\n12 - Pneumothorax\n\n13 - Pulmonary fibrosis\n\n14 - No finding","metadata":{}},{"cell_type":"markdown","source":"**Columns**\n\n**image_id** - unique image identifier\n\n**class_name** - the name of the class of detected object (or \"No finding\")\n\n\n**class_id** - the ID of the class of detected object\n\n**rad_id** - the ID of the radiologist that made the observation\n\n**x_min** - minimum X coordinate of the object's bounding box\n\n**y_min** - minimum Y coordinate of the object's bounding box\n\n**x_max** - maximum X coordinate of the object's bounding box\n\n**y_max** - maximum Y coordinate of the object's bounding box","metadata":{}},{"cell_type":"code","source":"def read_xray(path, voi_lut = True, fix_monochrome = True):\n    dicom = pydicom.read_file(path)\n    if voi_lut:\n        #apply voi lookup table if possible\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n        \n    #fix monochrome issue\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data.astype(np.float)\n    \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    #modified pixel range on a greyscale (0-255)    \n    return data","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Examples of diseases detection","metadata":{}},{"cell_type":"code","source":"def get_rect_patch(x_min,y_min,x_max,y_max):\n    width=x_max-x_min\n    height=y_max-y_min\n    rect=patches.Rectangle((x_min,y_min),width,height,ec='r', fc='none', lw=2.)\n    return rect","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for classname in train.class_name.unique():\n    fig,axs=plt.subplots(1,5,figsize=(20,15))\n    fig.subplots_adjust(hspace = .1, wspace=.1)\n\n    axs = axs.ravel()\n    \n    #sampling 3 images with corresponding disease\n    samples=train[train['class_name'] == classname].sample(n=5,random_state=353)\n    for i in range(5):\n        axs[i].imshow(read_xray(PATH+\"images/train/\"+samples.iloc[i]['image_id']+\".dicom\"),cmap='gray')\n\n        axs[i].set_title(str(classname))\n        axs[i].set_xticklabels([])\n        axs[i].set_yticklabels([])\n        \n        if classname != \"No finding\":\n            x_min,y_min,x_max,y_max=samples.iloc[i]['x_min'],samples.iloc[i]['y_min'],samples.iloc[i]['x_max'],samples.iloc[i]['y_max']\n            rect=get_rect_patch(x_min,y_min,x_max,y_max)\n            \n            axs[i].add_patch(rect)\nplt.show()","metadata":{"_kg_hide-output":false},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**YOLO Format** is as follows:\n* One row per object\n* Each row is class x_center y_center width height format.\n* Box coordinates must be in normalized xywh format (from 0 - 1). If your boxes are in pixels, divide x_center and width by image width, and y_center and height by image height.\n* Class numbers are zero-indexed (start from 0).","metadata":{}},{"cell_type":"code","source":"#Normalizing Data\ndef convert(size, box):\n    #xmin,xmax,ymin,ymax\n    dw = 1./(size[0])\n    dh = 1./(size[1])\n    x = (box[0] + box[1])/2.0 - 1\n    y = (box[2] + box[3])/2.0 - 1\n    w = box[1] - box[0]\n    h = box[3] - box[2]\n    x = x*dw\n    w = w*dw\n    y = y*dh\n    h = h*dh\n    return (x,y,w,h)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_to_yolo(df,IsTrain=True):\n    folder=\"train/\" if IsTrain else \"test/\"\n    global PATH\n    \n    r=pd.DataFrame(index=range(df.shape[0]),columns=['image_id','class','x','y','w','h','img_width','img_height'])\n    for i,img_id in enumerate()\n        print(str(i)+\"/\"+str(df.shape[0])+\"\\n\")\n        for idx,row in df.iterrows():\n            box=[row['x_min'],row['x_max'],row['y_min'],row['y_max']]\n            size=pydicom.read_file(PATH+\"images/train/\"+row['image_id']+\".dicom\").pixel_array.shape[::-1] \n            x,y,w,h=convert(size,box)\n            r.iloc[idx]['img_width'],r.iloc[idx]['img_height'],r.iloc[idx]['image_id'],r.iloc[idx]['class'],r.iloc[idx]['x'],r.iloc[idx]['y'],r.iloc[idx]['w'],r.iloc[idx]['h'] = size[0],size[1],row['image_id'],row['class_id'],x,y,w,h\n            gc.collect()\n            i+=1\n    return r","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r=convert_to_yolo(train)\nr.to_csv('chest_xray.csv')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r['image_id'].nunique()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicom_imgs_path=PATH+\"train/\"\n\ntrain_imgs_path=PATH+\"images/train/\"\ntrain_labels_path=PATH+\"labels/train/\"\n\ntest_imgs_path=PATH+\"images/test/\"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_classes=14","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Creating .txt files in yolov5 format\nfor i in r.image_id.unique():\n    img=read_xray(PATH+train_imgs_path+i+\".dicom\")\n    cv2.imwrite(\"./\" + train_imgs_path + i + \".jpg\" , img)\n    with open(PATH+train_labels_path+i+\".txt\",\"a\") as f:\n        f.seek(0)\n        img_df=r[r[\"image_id\"]== i]\n        found=False\n        nf_idxs=[]\n        for idx,row in img_df.iterrows():\n            if int(row['class']) == 14:\n                nf_idxs += idx\n                continue\n            else:\n                found=True\n                f.write(str(row['class']) + \" \" + str(row['x']) + \" \" + str(row[\"y\"]) + \" \" + str(row['w']) + \" \" + str(row['h']) + \"\\n\")\n        if found:\n            for idx in nf_idxs:\n                cv2.imwrite( \"./\" + train_imgs_path + i + \"_NF_\" + str(idx) + \".jpg\" , img)\n                f = open( PATH + train_labels_path + i + \"_NF_\" + str(idx) + \".txt\" , \"x\" )\n        f.close()","metadata":{"scrolled":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Converting test images into JPG\nfor i in sub.image_id.unique():\n    img=read_xray(PATH+test_imgs_path+i+\".dicom\")\n    cv2.imwrite(\"./\" + test_imgs_path + i + \".jpg\" , img)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test=sub.copy()\ntest['width']=test['height']=0\nfor idx,row in test.iterrows():\n    img = Image.open(PATH+test_imgs_path+row['image_id']+\".jpg\")\n    test.loc[idx,'width'],test.loc[idx,'height']=img.size[0],img.size[1]","metadata":{"scrolled":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"At this point i trained the model on a gpu cloud computing service ,namely : https://lambdalabs.com/.\nFor faster training i used : \n> python train.py torch.distributed.launch --nproc_per_node 4 ***etc...***","metadata":{}},{"cell_type":"markdown","source":"note that this notebook has to be adapted to the working dir to be able to run,as i didn't prepare it for kaggle deployement","metadata":{}},{"cell_type":"code","source":"os.chdir(f\"{PATH}yolov5-master\")\n!python train.py --weights \"./weights/yolov5x.pt\" --data ../chest_xray.yaml --epochs 60","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python detect.py --weights \"./runs/train/exp4/weights/best.pt\" --save-txt --save-conf --sources \"../images/test/\" --img-size 640","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def yolo2voc(image_height, image_width, bboxes):\n    \"\"\"\n    yolo => [xmid, ymid, w, h] (normalized)\n    voc  => [x1, y1, x2, y1]\n    \n    \"\"\" \n    bboxes = bboxes.copy().astype(float) # otherwise all value will be 0 as voc_pascal dtype is np.int\n    \n    bboxes[..., [0, 2]] = bboxes[..., [0, 2]]* image_width\n    bboxes[..., [1, 3]] = bboxes[..., [1, 3]]* image_height\n    \n    bboxes[..., [0, 1]] = bboxes[..., [0, 1]] - bboxes[..., [2, 3]]/2\n    bboxes[..., [2, 3]] = bboxes[..., [0, 1]] + bboxes[..., [2, 3]]\n    \n    return bboxes","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_ids = []\nPredictionStrings = []\n\nfor file_path in glob('runs/detect/exp/labels/*txt'):\n    image_id = file_path.split('/')[-1].split('.')[0].split(\"\\\\\")[1]\n    w, h = int(test.loc[test.image_id==image_id]['width']),int(test.loc[test.image_id==image_id]['height'])\n    f = open(file_path, 'r')\n    data = np.array(f.read().replace('\\n', ' ').strip().split(' ')).astype(np.float32).reshape(-1, 6)\n    data = data[:, [0, 5, 1, 2, 3, 4]]\n    bboxes = list(np.round(np.concatenate((data[:, :2], np.round(yolo2voc(h, w, data[:, 2:]))), axis =1).reshape(-1), 1).astype(str))\n    for idx in range(len(bboxes)):\n        bboxes[idx] = str(int(float(bboxes[idx]))) if idx%6!=1 else bboxes[idx]\n    image_ids.append(image_id)\n    PredictionStrings.append(' '.join(bboxes))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df=test.copy().drop(\"PredictionString\",axis=1,inplace=False)\npred_df = pd.DataFrame({'image_id':image_ids,\n                        'PredictionString':PredictionStrings})\nsub_df = pd.merge(test_df, pred_df, on = 'image_id', how = 'left').fillna(\"14 1 0 0 1 1\")\nprint(sub_df)\nsub_df = sub_df.loc[:,['image_id', 'PredictionString']]\nsub_df.to_csv('submission.csv',index = False)\nsub_df.tail()","metadata":{},"execution_count":null,"outputs":[]}]}