{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Hi everyone, \n\nIn this notebook, I would like to show you how we can extract the breast only part from the input images by using OpenCV and [Otsu's thresholding](https://en.wikipedia.org/wiki/Otsu%27s_method).\nSource code of this repository can be found here: https://github.com/Guarouba/breast_mass_detection\n\nI am using [@radek1](https://www.kaggle.com/radek1)'s https://www.kaggle.com/competitions/rsna-breast-cancer-detection/discussion/369762 processed dataset.\n\nLet's start with the imports:","metadata":{}},{"cell_type":"code","source":"import glob\nimport cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport pandas as pd\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-12-19T09:13:51.236821Z","iopub.execute_input":"2022-12-19T09:13:51.237162Z","iopub.status.idle":"2022-12-19T09:13:51.403729Z","shell.execute_reply.started":"2022-12-19T09:13:51.237095Z","shell.execute_reply":"2022-12-19T09:13:51.402726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/mmcvfull/mmcvfull/addict-2.4.0-py3-none-any.whl\n!pip install /kaggle/input/mmcvfull/mmcvfull/mmcv_full-1.7.0-cp37-cp37m-manylinux1_x86_64.whl\n!pip install /kaggle/input/mmmcls/mmcls-0.25.0-py2.py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2022-12-19T09:13:56.527197Z","iopub.execute_input":"2022-12-19T09:13:56.527551Z","iopub.status.idle":"2022-12-19T09:15:28.792125Z","shell.execute_reply.started":"2022-12-19T09:13:56.52752Z","shell.execute_reply":"2022-12-19T09:15:28.79096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/onnxruntime/onnx/humanfriendly-10.0-py2.py3-none-any.whl\n!pip install /kaggle/input/onnxruntime/onnx/coloredlogs-15.0.1-py2.py3-none-any.whl\n!pip install /kaggle/input/onnxruntime/onnx/onnxruntime-1.13.1-cp37-cp37m-manylinux_2_27_x86_64.whl","metadata":{"execution":{"iopub.status.busy":"2022-12-19T09:16:36.998392Z","iopub.execute_input":"2022-12-19T09:16:36.99879Z","iopub.status.idle":"2022-12-19T09:18:06.394186Z","shell.execute_reply.started":"2022-12-19T09:16:36.998757Z","shell.execute_reply":"2022-12-19T09:18:06.39303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import onnxruntime","metadata":{"execution":{"iopub.status.busy":"2022-12-19T09:18:15.018701Z","iopub.execute_input":"2022-12-19T09:18:15.01976Z","iopub.status.idle":"2022-12-19T09:18:15.048611Z","shell.execute_reply.started":"2022-12-19T09:18:15.019711Z","shell.execute_reply":"2022-12-19T09:18:15.047744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip3 install openmim --target=/kaggle/working/\n# !mim install mmcv-full --target=/kaggle/working/\n# !mim install mmcls\n# !pip3 install -r /kaggle/input/mmclassification/mmclassification-master/requirements.txt\n# !pip3 install -r /kaggle/input/mmclassification/mmclassification-master/requirements/optional.txt\n# !pip3 install -qU python-gdcm pydicom pylibjpeg\n# !pip3 install onnxruntime","metadata":{"execution":{"iopub.status.busy":"2022-12-19T09:18:19.251773Z","iopub.execute_input":"2022-12-19T09:18:19.252156Z","iopub.status.idle":"2022-12-19T09:18:19.256703Z","shell.execute_reply.started":"2022-12-19T09:18:19.252122Z","shell.execute_reply":"2022-12-19T09:18:19.255508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#reverse image\nimport pydicom\ndef dcm2png(path):\n    dicom = pydicom.dcmread(path)\n    img = dicom.pixel_array\n\n    img = (img - img.min()) / (img.max() - img.min())\n\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n\n    img = cv2.resize(img, (1024, 1024))\n    img = (img * 255).astype(np.uint8)\n    img = np.array([img,img,img]).transpose(1,2,0)\n    return img","metadata":{"execution":{"iopub.status.busy":"2022-12-19T09:18:21.184856Z","iopub.execute_input":"2022-12-19T09:18:21.18588Z","iopub.status.idle":"2022-12-19T09:18:21.296346Z","shell.execute_reply.started":"2022-12-19T09:18:21.185843Z","shell.execute_reply":"2022-12-19T09:18:21.295417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#detect image\nimport os\nimport cv2\nimport numpy as np\nimport time\n\nCLASSES=['breast'] #coco80类别\n\nclass YOLOV5():\n    def __init__(self,onnxpath):\n        self.onnx_session=onnxruntime.InferenceSession(onnxpath)\n        self.input_name=self.get_input_name()\n        self.output_name=self.get_output_name()\n    #-------------------------------------------------------\n\t#   获取输入输出的名字\n\t#-------------------------------------------------------\n    def get_input_name(self):\n        input_name=[]\n        for node in self.onnx_session.get_inputs():\n            input_name.append(node.name)\n        return input_name\n    def get_output_name(self):\n        output_name=[]\n        for node in self.onnx_session.get_outputs():\n            output_name.append(node.name)\n        return output_name\n    #-------------------------------------------------------\n\t#   输入图像\n\t#-------------------------------------------------------\n    def get_input_feed(self,img_tensor):\n        input_feed={}\n        for name in self.input_name:\n            input_feed[name]=img_tensor\n        return input_feed\n    #-------------------------------------------------------\n\t#   1.cv2读取图像并resize\n\t#\t2.图像转BGR2RGB和HWC2CHW\n\t#\t3.图像归一化\n\t#\t4.图像增加维度\n\t#\t5.onnx_session 推理\n\t#-------------------------------------------------------\n    def inference(self,img):\n        or_img=cv2.resize(img,(640,640))\n        img=or_img[:,:,::-1].transpose(2,0,1)  #BGR2RGB和HWC2CHW\n        img=img.astype(dtype=np.float32)\n        img/=255.0\n        img=np.expand_dims(img,axis=0)\n        input_feed=self.get_input_feed(img)\n        pred=self.onnx_session.run(None,input_feed)[0]\n        return pred,or_img\n\n#dets:  array [x,6] 6个值分别为x1,y1,x2,y2,score,class \n#thresh: 阈值\ndef nms(dets, thresh):\n    x1 = dets[:, 0]\n    y1 = dets[:, 1]\n    x2 = dets[:, 2]\n    y2 = dets[:, 3]\n    #-------------------------------------------------------\n\t#   计算框的面积\n    #\t置信度从大到小排序\n\t#-------------------------------------------------------\n    areas = (y2 - y1 + 1) * (x2 - x1 + 1)\n    scores = dets[:, 4]\n    keep = []\n    index = scores.argsort()[::-1] \n\n    while index.size > 0:\n        i = index[0]\n        keep.append(i)\n\t\t#-------------------------------------------------------\n        #   计算相交面积\n        #\t1.相交\n        #\t2.不相交\n        #-------------------------------------------------------\n        x11 = np.maximum(x1[i], x1[index[1:]]) \n        y11 = np.maximum(y1[i], y1[index[1:]])\n        x22 = np.minimum(x2[i], x2[index[1:]])\n        y22 = np.minimum(y2[i], y2[index[1:]])\n\n        w = np.maximum(0, x22 - x11 + 1)                              \n        h = np.maximum(0, y22 - y11 + 1) \n\n        overlaps = w * h\n        #-------------------------------------------------------\n        #   计算该框与其它框的IOU，去除掉重复的框，即IOU值大的框\n        #\tIOU小于thresh的框保留下来\n        #-------------------------------------------------------\n        ious = overlaps / (areas[i] + areas[index[1:]] - overlaps)\n        idx = np.where(ious <= thresh)[0]\n        index = index[idx + 1]\n    return keep\n\n\ndef xywh2xyxy(x):\n    # [x, y, w, h] to [x1, y1, x2, y2]\n    y = np.copy(x)\n    y[:, 0] = x[:, 0] - x[:, 2] / 2\n    y[:, 1] = x[:, 1] - x[:, 3] / 2\n    y[:, 2] = x[:, 0] + x[:, 2] / 2\n    y[:, 3] = x[:, 1] + x[:, 3] / 2\n    return y\n\ndef save_maxarea(det):\n    MaxArea = []\n    #if len(det)>1:\n    for *xyxy, conf, cls in reversed(det): \n        w = xyxy[2] - xyxy[0]\n        h = xyxy[3] - xyxy[1]\n        area = w*h\n        MaxArea.append(-area)\n    #print(MaxArea)\n    Max = max(MaxArea)  \n    # 最大值\n    MaxI = MaxArea.index(Max)\n    det = det[[MaxI]]\n    return det\n\ndef filter_box(org_box,conf_thres,iou_thres): #过滤掉无用的框\n    #-------------------------------------------------------\n\t#   删除为1的维度\n    #\t删除置信度小于conf_thres的BOX\n\t#-------------------------------------------------------\n    org_box=np.squeeze(org_box)\n    conf = org_box[..., 4] > conf_thres\n    box = org_box[conf == True]\n    #-------------------------------------------------------\n    #\t通过argmax获取置信度最大的类别\n\t#-------------------------------------------------------\n    cls_cinf = box[..., 5:]\n    cls = []\n    for i in range(len(cls_cinf)):\n        cls.append(int(np.argmax(cls_cinf[i])))\n    all_cls = list(set(cls))     \n    #-------------------------------------------------------\n\t#   分别对每个类别进行过滤\n\t#\t1.将第6列元素替换为类别下标\n\t#\t2.xywh2xyxy 坐标转换\n\t#\t3.经过非极大抑制后输出的BOX下标\n\t#\t4.利用下标取出非极大抑制后的BOX\n\t#-------------------------------------------------------\n    output = []\n    for i in range(len(all_cls)):\n        curr_cls = all_cls[i]\n        curr_cls_box = []\n        curr_out_box = []\n        for j in range(len(cls)):\n            if cls[j] == curr_cls:\n                box[j][5] = curr_cls\n                curr_cls_box.append(box[j][:6])\n        curr_cls_box = np.array(curr_cls_box)\n        # curr_cls_box_old = np.copy(curr_cls_box)\n        curr_cls_box = xywh2xyxy(curr_cls_box)\n        curr_out_box = nms(curr_cls_box,iou_thres)\n        for k in curr_out_box:\n            output.append(curr_cls_box[k])\n    output = np.array(output)\n    return output\n\ndef clip_image(image,box_data):  \n    #-------------------------------------------------------\n    #\t取整，方便裁剪\n\t#-------------------------------------------------------\n    boxes=box_data[...,:4].astype(np.int32) \n    top, left, right, bottom = boxes[0]\n    image = image[np.clip(top,0,image.shape[0]):np.clip(bottom,0,image.shape[0]),np.clip(left,0,image.shape[1]):np.clip(right,0,image.shape[1]),:]\n    return image\n    \n\ndef draw(image,box_data):  \n    #-------------------------------------------------------\n    #\t取整，方便画框\n\t#-------------------------------------------------------\n    boxes=box_data[...,:4].astype(np.int32) \n    scores=box_data[...,4]\n    classes=box_data[...,5].astype(np.int32) \n\n    for box, score, cl in zip(boxes, scores, classes):\n        top, left, right, bottom = box\n#         print('class: {}, score: {}'.format(CLASSES[cl], score))\n#         print('box coordinate left,top,right,down: [{}, {}, {}, {}]'.format(top, left, right, bottom))\n\n        cv2.rectangle(image, (top, left), (right, bottom), (255, 0, 0), 2)\n        cv2.putText(image, '{0} {1:.2f}'.format(CLASSES[cl], score),\n                    (top, left ),\n                    cv2.FONT_HERSHEY_SIMPLEX,\n                    0.6, (0, 0, 255), 2)","metadata":{"execution":{"iopub.status.busy":"2022-12-19T09:18:24.741349Z","iopub.execute_input":"2022-12-19T09:18:24.741745Z","iopub.status.idle":"2022-12-19T09:18:24.771918Z","shell.execute_reply.started":"2022-12-19T09:18:24.741712Z","shell.execute_reply":"2022-12-19T09:18:24.770851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#clip image\nonnx_path='/kaggle/input/mmclassification/mmclassification-master/work_dirs/best.onnx'\nmodel_det=YOLOV5(onnx_path)\ndef Clip_image(model,image):\n    output,or_img=model.inference(image)\n    outbox=filter_box(output,0.3,0.3)\n    outbox = save_maxarea(outbox)\n    or_img = clip_image(or_img,outbox)\n    return or_img","metadata":{"execution":{"iopub.status.busy":"2022-12-19T09:18:28.151066Z","iopub.execute_input":"2022-12-19T09:18:28.152051Z","iopub.status.idle":"2022-12-19T09:18:29.095698Z","shell.execute_reply.started":"2022-12-19T09:18:28.152006Z","shell.execute_reply":"2022-12-19T09:18:29.094611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_detail(name):\n    models = {\n        'breast': {\n            'config':\n            '/kaggle/input/mmclassification/mmclassification-master/work_dirs/convnext_large_cancer/convnext_large_cancer.py',\n            'ckpt':\n            '/kaggle/input/mmclassification/mmclassification-master/work_dirs/convnext_large_cancer/best_accuracy_top-1_epoch_5.pth'\n        }\n    }\n    return models[name]['config'],models[name]['ckpt']","metadata":{"execution":{"iopub.status.busy":"2022-12-19T09:18:30.856849Z","iopub.execute_input":"2022-12-19T09:18:30.857223Z","iopub.status.idle":"2022-12-19T09:18:30.862604Z","shell.execute_reply.started":"2022-12-19T09:18:30.857191Z","shell.execute_reply":"2022-12-19T09:18:30.861545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nimport os\nimport cv2\nimport mmcv\nimport shutil\nimport numpy as np\nimport albumentations as A\nfrom mmcls.apis import inference_model, init_model, show_result_pyplot","metadata":{"execution":{"iopub.status.busy":"2022-12-19T09:18:32.588399Z","iopub.execute_input":"2022-12-19T09:18:32.588773Z","iopub.status.idle":"2022-12-19T09:18:37.537557Z","shell.execute_reply.started":"2022-12-19T09:18:32.588742Z","shell.execute_reply":"2022-12-19T09:18:37.536495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config,check = get_detail('breast')\nmodel_cls = init_model(config, check, device='cuda')","metadata":{"execution":{"iopub.status.busy":"2022-12-19T09:18:47.430996Z","iopub.execute_input":"2022-12-19T09:18:47.432202Z","iopub.status.idle":"2022-12-19T09:19:17.212645Z","shell.execute_reply.started":"2022-12-19T09:18:47.432155Z","shell.execute_reply":"2022-12-19T09:19:17.210766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imagedir = \"/kaggle/input/rsna-breast-cancer-detection/test_images/\"","metadata":{"execution":{"iopub.status.busy":"2022-12-19T09:19:39.871391Z","iopub.execute_input":"2022-12-19T09:19:39.871774Z","iopub.status.idle":"2022-12-19T09:19:39.876769Z","shell.execute_reply.started":"2022-12-19T09:19:39.871741Z","shell.execute_reply":"2022-12-19T09:19:39.875813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testcsvdir = \"/kaggle/input/rsna-breast-cancer-detection/test.csv\"","metadata":{"execution":{"iopub.status.busy":"2022-12-19T09:20:22.706412Z","iopub.execute_input":"2022-12-19T09:20:22.706778Z","iopub.status.idle":"2022-12-19T09:20:22.711477Z","shell.execute_reply.started":"2022-12-19T09:20:22.706749Z","shell.execute_reply":"2022-12-19T09:20:22.710486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_ids = []\npreds = []","metadata":{"execution":{"iopub.status.busy":"2022-12-19T09:20:24.424156Z","iopub.execute_input":"2022-12-19T09:20:24.424508Z","iopub.status.idle":"2022-12-19T09:20:24.429438Z","shell.execute_reply.started":"2022-12-19T09:20:24.424478Z","shell.execute_reply":"2022-12-19T09:20:24.428354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_csv = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\npatient_id = np.unique(test_csv['patient_id'].values.tolist()).tolist()\nfor patient in patient_id:\n    patient_csv = test_csv[test_csv['patient_id']==patient]\n    laterality_list = np.unique(patient_csv['laterality'].values.tolist()).tolist()\n    for laterality in laterality_list:\n        image_id_csv = patient_csv[patient_csv['laterality']==laterality]\n        image_id_list = np.unique(image_id_csv['image_id'].values.tolist()).tolist()\n        resname = str(patient)+\"_\"+laterality\n        total_score = np.zeros(2)\n        total_imgnum = np.zeros(2)\n        for image_id in image_id_list:\n            imagepath = imagedir+str(patient)+\"/\"+str(image_id)+\".dcm\"\n            try:\n                image = dcm2png(imagepath)\n            except:\n                prediction_ids.append(resname)\n                preds.append(0)\n                continue\n            try:\n                clipimage = Clip_image(model_det,image)\n            except:\n                prediction_ids.append(resname)\n                preds.append(0)\n                continue\n#             plt.imshow(clipimage)\n#             plt.show()\n            try:\n                result = inference_model(model_cls, clipimage)\n            except:\n                prediction_ids.append(resname)\n                preds.append(0)\n                continue\n            total_score[int(result['pred_label'])] += float(result['pred_score'])\n            total_imgnum[int(result['pred_label'])] += 1\n        index  = np.argmax(total_score)\n        if index == 1:\n            prediction_ids.append(resname)\n            preds.append(0)\n        else:\n            prediction_ids.append(resname)\n            preds.append(1)\nsubmission = pd.DataFrame(data={'prediction_id': prediction_ids, 'cancer': preds}).groupby('prediction_id').max().reset_index()\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-19T09:20:26.87478Z","iopub.execute_input":"2022-12-19T09:20:26.875479Z","iopub.status.idle":"2022-12-19T09:20:37.803849Z","shell.execute_reply.started":"2022-12-19T09:20:26.875431Z","shell.execute_reply":"2022-12-19T09:20:37.802939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-19T04:47:40.843732Z","iopub.execute_input":"2022-12-19T04:47:40.844095Z","iopub.status.idle":"2022-12-19T04:47:40.851003Z","shell.execute_reply.started":"2022-12-19T04:47:40.844063Z","shell.execute_reply":"2022-12-19T04:47:40.849699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}