{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"raw","source":"##### Yolov5 Cervical Spine (Neck) Fracture Detection","metadata":{"papermill":{"duration":0.096947,"end_time":"2021-07-18T05:37:05.065223","exception":false,"start_time":"2021-07-18T05:37:04.968276","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T08:00:38.103499Z","iopub.execute_input":"2022-08-06T08:00:38.103969Z","iopub.status.idle":"2022-08-06T08:00:38.109681Z","shell.execute_reply.started":"2022-08-06T08:00:38.103929Z","shell.execute_reply":"2022-08-06T08:00:38.108025Z"}}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\n\n\nimport cv2\nimport time\nimport shutil\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\n\nimport torchvision\nimport torchvision.transforms as transforms\n\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms, utils\n\nfrom torch.utils.data import Dataset, DataLoader\n\nimport tqdm.notebook as tq\n\nimport gc\n\n\nimport albumentations as albu\nfrom albumentations import Compose\n\n\nfrom sklearn import model_selection\nfrom sklearn.utils import shuffle\nfrom sklearn import metrics\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import classification_report, jaccard_score\nimport itertools\n\nfrom PIL import Image\nfrom numpy import asarray\nfrom skimage.transform import resize\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings('ignore')\nprint(torch.__version__)\nprint(torchvision.__version__)","metadata":{"papermill":{"duration":11.55616,"end_time":"2021-07-18T05:38:38.821538","exception":false,"start_time":"2021-07-18T05:38:27.265378","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:35.287297Z","iopub.execute_input":"2022-10-25T08:19:35.287802Z","iopub.status.idle":"2022-10-25T08:19:39.596787Z","shell.execute_reply.started":"2022-10-25T08:19:35.287692Z","shell.execute_reply":"2022-10-25T08:19:39.595032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\nseed_val = 101\nos.environ['PYTHONHASHSEED'] = str(seed_val)\nrandom.seed(seed_val)\nnp.random.seed(seed_val)\ntorch.manual_seed(seed_val)\ntorch.cuda.manual_seed_all(seed_val)\ntorch.backends.cudnn.deterministic = True","metadata":{"papermill":{"duration":0.227737,"end_time":"2021-07-18T05:38:39.265366","exception":false,"start_time":"2021-07-18T05:38:39.037629","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:39.598974Z","iopub.execute_input":"2022-10-25T08:19:39.599536Z","iopub.status.idle":"2022-10-25T08:19:39.609469Z","shell.execute_reply.started":"2022-10-25T08:19:39.599499Z","shell.execute_reply":"2022-10-25T08:19:39.608538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('../input/')","metadata":{"papermill":{"duration":0.22636,"end_time":"2021-07-18T05:38:39.705411","exception":false,"start_time":"2021-07-18T05:38:39.479051","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:39.61108Z","iopub.execute_input":"2022-10-25T08:19:39.611498Z","iopub.status.idle":"2022-10-25T08:19:39.624052Z","shell.execute_reply.started":"2022-10-25T08:19:39.611462Z","shell.execute_reply":"2022-10-25T08:19:39.623127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_path = '../input/rsna-2022-cervical-spine-fracture-detection/'\n\nprep_data_path = '../input/spine-fracture-comp-data/'\n","metadata":{"papermill":{"duration":0.220316,"end_time":"2021-07-18T05:38:40.140516","exception":false,"start_time":"2021-07-18T05:38:39.9202","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:39.626776Z","iopub.execute_input":"2022-10-25T08:19:39.627292Z","iopub.status.idle":"2022-10-25T08:19:39.631855Z","shell.execute_reply.started":"2022-10-25T08:19:39.627259Z","shell.execute_reply":"2022-10-25T08:19:39.630685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Config","metadata":{}},{"cell_type":"code","source":"# Yolo setup:\nNUM_EPOCHS = 80\nBATCH_SIZE = 32\nIMAGE_SIZE = 512 # Yolo will automatically resize the input images to this size.\n\n# This is the fold that Yolo is trained on\nCHOSEN_FOLD = 0\n\nNUM_FOLDS = 5\n\nNUM_CORES = os.cpu_count()\nNUM_CORES","metadata":{"papermill":{"duration":0.225834,"end_time":"2021-07-18T05:38:40.577433","exception":false,"start_time":"2021-07-18T05:38:40.351599","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:39.633136Z","iopub.execute_input":"2022-10-25T08:19:39.63389Z","iopub.status.idle":"2022-10-25T08:19:39.644263Z","shell.execute_reply.started":"2022-10-25T08:19:39.633856Z","shell.execute_reply":"2022-10-25T08:19:39.643244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Set up Yolov5 - For offline use\n\nThe Yolov5 model being used here needs to have the internet turned for training to work. However, it does not need to have the internet on during inference.","metadata":{"papermill":{"duration":0.215149,"end_time":"2021-07-18T05:38:41.008781","exception":false,"start_time":"2021-07-18T05:38:40.793632","status":"completed"},"tags":[]}},{"cell_type":"code","source":"shutil.copytree('../input/v2-balloon-detection-dataset/yolov5', '/kaggle/working/yolov5')","metadata":{"papermill":{"duration":1.303094,"end_time":"2021-07-18T05:38:42.525305","exception":false,"start_time":"2021-07-18T05:38:41.222211","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:39.645834Z","iopub.execute_input":"2022-10-25T08:19:39.646388Z","iopub.status.idle":"2022-10-25T08:19:40.312045Z","shell.execute_reply.started":"2022-10-25T08:19:39.646355Z","shell.execute_reply":"2022-10-25T08:19:40.311132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shutil.copyfile('../input/finalmod/best.pt', '/kaggle/working/best.pt')","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:19:40.314364Z","iopub.execute_input":"2022-10-25T08:19:40.315245Z","iopub.status.idle":"2022-10-25T08:19:41.449508Z","shell.execute_reply.started":"2022-10-25T08:19:40.315205Z","shell.execute_reply":"2022-10-25T08:19:41.448611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shutil.copyfile('../input/rsna-2022-cervical-spine-fracture-detection/test_images/1.2.826.0.1.3680043.22327/100.dcm', '/kaggle/working/100.dcm')","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:19:41.450783Z","iopub.execute_input":"2022-10-25T08:19:41.451139Z","iopub.status.idle":"2022-10-25T08:19:41.470509Z","shell.execute_reply.started":"2022-10-25T08:19:41.451103Z","shell.execute_reply":"2022-10-25T08:19:41.469504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"papermill":{"duration":1.013171,"end_time":"2021-07-18T05:38:43.756671","exception":false,"start_time":"2021-07-18T05:38:42.7435","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:41.472164Z","iopub.execute_input":"2022-10-25T08:19:41.472528Z","iopub.status.idle":"2022-10-25T08:19:42.416991Z","shell.execute_reply.started":"2022-10-25T08:19:41.472494Z","shell.execute_reply":"2022-10-25T08:19:42.415792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load the train data","metadata":{"papermill":{"duration":0.215286,"end_time":"2021-07-18T05:38:44.189361","exception":false,"start_time":"2021-07-18T05:38:43.974075","status":"completed"},"tags":[]}},{"cell_type":"code","source":"path = prep_data_path + 'df_data.csv'\n\ndf_data = pd.read_csv(path)\n\nprint(df_data.shape)\n\ndf_data.head()","metadata":{"papermill":{"duration":0.415195,"end_time":"2021-07-18T05:38:44.823811","exception":false,"start_time":"2021-07-18T05:38:44.408616","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:42.421318Z","iopub.execute_input":"2022-10-25T08:19:42.421723Z","iopub.status.idle":"2022-10-25T08:19:42.49145Z","shell.execute_reply.started":"2022-10-25T08:19:42.42169Z","shell.execute_reply":"2022-10-25T08:19:42.49057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:19:42.492795Z","iopub.execute_input":"2022-10-25T08:19:42.493173Z","iopub.status.idle":"2022-10-25T08:19:42.506343Z","shell.execute_reply.started":"2022-10-25T08:19:42.493144Z","shell.execute_reply":"2022-10-25T08:19:42.505223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Process the train data","metadata":{"papermill":{"duration":0.217084,"end_time":"2021-07-18T05:38:45.726327","exception":false,"start_time":"2021-07-18T05:38:45.509243","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Create a column called 'target'\n\ndf_data['target'] = list(df_data['label'])\n\n#df_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:19:42.508212Z","iopub.execute_input":"2022-10-25T08:19:42.5091Z","iopub.status.idle":"2022-10-25T08:19:42.51847Z","shell.execute_reply.started":"2022-10-25T08:19:42.509064Z","shell.execute_reply":"2022-10-25T08:19:42.517471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Targets:\n# 0 = normal\n# 1 = fracture\n\ndf_data['target'].value_counts()","metadata":{"papermill":{"duration":0.231376,"end_time":"2021-07-18T05:38:47.246869","exception":false,"start_time":"2021-07-18T05:38:47.015493","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:42.520258Z","iopub.execute_input":"2022-10-25T08:19:42.520633Z","iopub.status.idle":"2022-10-25T08:19:42.529125Z","shell.execute_reply.started":"2022-10-25T08:19:42.520599Z","shell.execute_reply":"2022-10-25T08:19:42.528058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create a column for the bbox info","metadata":{}},{"cell_type":"code","source":"# Load the bbox data\n\n# This dataframe lists all slides that have fractures\n\npath = base_path + 'train_bounding_boxes.csv'\ndf_bbox = pd.read_csv(path)\n\nprint(df_bbox.shape)\n\ndf_bbox.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:19:42.530818Z","iopub.execute_input":"2022-10-25T08:19:42.531281Z","iopub.status.idle":"2022-10-25T08:19:42.566243Z","shell.execute_reply.started":"2022-10-25T08:19:42.531241Z","shell.execute_reply":"2022-10-25T08:19:42.565304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_study_slice(row):\n    \n    study_id = str(row['StudyInstanceUID'])\n    slice_num = str(row['slice_number'])\n    \n    study_slice = study_id + '_' + slice_num\n    \n    return study_slice\n\ndf_bbox['study_slice'] = df_bbox.apply(create_study_slice, axis=1)\n\ndf_bbox = df_bbox.set_index('study_slice')\n\nprint(df_bbox.shape)\n\ndf_bbox.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:19:42.567589Z","iopub.execute_input":"2022-10-25T08:19:42.568389Z","iopub.status.idle":"2022-10-25T08:19:42.683778Z","shell.execute_reply.started":"2022-10-25T08:19:42.568355Z","shell.execute_reply":"2022-10-25T08:19:42.682835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Put the bbox info for each image into a list of bbox dicts\n\n# Note: According to df_bbox we only have one bbox per image (slice)\n\nstudy_slice_list = list(df_data['study_slice'])\n\nbbox_list = []\n\nfor i in range(0, len(df_data)):\n    \n    target = df_data.loc[i, 'target']\n    study_slice = df_data.loc[i, 'study_slice']\n    \n    if target == 1:\n    \n        x = df_bbox.loc[study_slice, 'x']\n        y = df_bbox.loc[study_slice, 'y']\n        width = df_bbox.loc[study_slice, 'width']\n        height = df_bbox.loc[study_slice, 'height']\n\n        bbox_dict ={\n            'x': x,\n            'y': y,\n            'width': width,\n            'height': height\n        }\n\n        bbox_list.append(bbox_dict)\n        \n    else:\n        bbox_list.append('none')\n        \n        \n# Add the bbox_list to df_data\n\ndf_data['boxes'] = bbox_list\n\n#print(df_data.shape)\n\n#df_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:19:42.685369Z","iopub.execute_input":"2022-10-25T08:19:42.686022Z","iopub.status.idle":"2022-10-25T08:19:43.105824Z","shell.execute_reply.started":"2022-10-25T08:19:42.685984Z","shell.execute_reply":"2022-10-25T08:19:43.104838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:19:43.108617Z","iopub.execute_input":"2022-10-25T08:19:43.109227Z","iopub.status.idle":"2022-10-25T08:19:43.124537Z","shell.execute_reply.started":"2022-10-25T08:19:43.109191Z","shell.execute_reply":"2022-10-25T08:19:43.123609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display one entry\n\ndf_data.loc[0, 'boxes']","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:19:43.413686Z","iopub.execute_input":"2022-10-25T08:19:43.41405Z","iopub.status.idle":"2022-10-25T08:19:43.420851Z","shell.execute_reply.started":"2022-10-25T08:19:43.41402Z","shell.execute_reply":"2022-10-25T08:19:43.419869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Helper functions","metadata":{"papermill":{"duration":0.401781,"end_time":"2021-07-18T05:38:47.878685","exception":false,"start_time":"2021-07-18T05:38:47.476904","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def plot_confusion_matrix(cm, classes,\n                          normalize=False,\n                          title='Confusion matrix',\n                          cmap=plt.cm.Blues,\n                         text_size=12):\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n        print(\"Normalized confusion matrix\")\n    else:\n        print('Confusion matrix, without normalization')\n\n    print(cm)\n\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45, fontsize=text_size)\n    plt.yticks(tick_marks, classes, fontsize=text_size)\n\n    fmt = '.2f' if normalize else 'd'\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, format(cm[i, j], fmt),\n                 horizontalalignment=\"center\",\n                 color=\"white\" if cm[i, j] > thresh else \"black\", fontsize=text_size)\n\n    plt.ylabel('True label', fontsize=text_size)\n    plt.xlabel('Predicted label', fontsize=text_size)\n    plt.tight_layout()","metadata":{"papermill":{"duration":0.227809,"end_time":"2021-07-18T05:38:48.408535","exception":false,"start_time":"2021-07-18T05:38:48.180726","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:44.781095Z","iopub.execute_input":"2022-10-25T08:19:44.781449Z","iopub.status.idle":"2022-10-25T08:19:44.792018Z","shell.execute_reply.started":"2022-10-25T08:19:44.78142Z","shell.execute_reply":"2022-10-25T08:19:44.790836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\n\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    dicom = pydicom.read_file(path)\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n            \n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data\n\n\n\n\ndef resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    \n    return im","metadata":{"papermill":{"duration":0.411458,"end_time":"2021-07-18T05:38:50.75773","exception":false,"start_time":"2021-07-18T05:38:50.346272","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:45.123959Z","iopub.execute_input":"2022-10-25T08:19:45.124334Z","iopub.status.idle":"2022-10-25T08:19:45.233886Z","shell.execute_reply.started":"2022-10-25T08:19:45.124303Z","shell.execute_reply":"2022-10-25T08:19:45.232975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create the folds","metadata":{"papermill":{"duration":0.215665,"end_time":"2021-07-18T05:38:57.522433","exception":false,"start_time":"2021-07-18T05:38:57.306768","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from sklearn.model_selection import KFold, StratifiedKFold\n\nskf = StratifiedKFold(n_splits=NUM_FOLDS, shuffle=True, random_state=101)\n\nfor fold, ( _, val_) in enumerate(skf.split(X=df_data, y=df_data.target)):\n      df_data.loc[val_ , \"fold\"] = fold\n        \ndf_data['fold'].value_counts()","metadata":{"papermill":{"duration":1.022182,"end_time":"2021-07-18T05:38:58.751994","exception":false,"start_time":"2021-07-18T05:38:57.729812","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:45.995751Z","iopub.execute_input":"2022-10-25T08:19:45.996776Z","iopub.status.idle":"2022-10-25T08:19:46.018734Z","shell.execute_reply.started":"2022-10-25T08:19:45.996729Z","shell.execute_reply":"2022-10-25T08:19:46.01783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the target distribution in each fold\n\nfor fold_index in range(0, NUM_FOLDS):\n    \n    df_train = df_data[df_data['fold'] != fold_index]\n    df_val = df_data[df_data['fold'] == fold_index]\n\n    print(f'\\nFold {fold_index}')\n    print('.........')\n    print()\n    print('Train shape:',df_train.shape)\n    print('Val shape:',df_val.shape)\n    print()\n    print('Train target distribution')\n    print(df_train['target'].value_counts())\n    print()\n    print('Val target distribution')\n    print(df_val['target'].value_counts())","metadata":{"papermill":{"duration":0.254228,"end_time":"2021-07-18T05:38:59.825215","exception":false,"start_time":"2021-07-18T05:38:59.570987","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:46.468933Z","iopub.execute_input":"2022-10-25T08:19:46.47091Z","iopub.status.idle":"2022-10-25T08:19:46.499833Z","shell.execute_reply.started":"2022-10-25T08:19:46.470855Z","shell.execute_reply":"2022-10-25T08:19:46.498743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create the Yolo directory structure\n\nWe need to create a directory structure inside the yolov5 folder. This is where the training and validation data will need to be stored","metadata":{"papermill":{"duration":0.214448,"end_time":"2021-07-18T05:39:00.717004","exception":false,"start_time":"2021-07-18T05:39:00.502556","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Note the the following folder structure must be\n# located inside the yolov5 folder\n\n# base_dir\n    # images\n        # train (contains image files)\n        # validation (contains image files)\n    # labels \n        # train (contains .txt files)\n        # validation (contains .txt files)\n        \n# Yolo expects the bounding box dimensions to be\n# normalized to have values between 0 and 1.\n        \n# Label format in .txt file\n# class x-center y-center width height\n# E.g. 0 0.1 0.2 200 300","metadata":{"papermill":{"duration":0.241002,"end_time":"2021-07-18T05:39:01.171558","exception":false,"start_time":"2021-07-18T05:39:00.930556","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:47.701872Z","iopub.execute_input":"2022-10-25T08:19:47.702241Z","iopub.status.idle":"2022-10-25T08:19:47.707048Z","shell.execute_reply.started":"2022-10-25T08:19:47.70221Z","shell.execute_reply":"2022-10-25T08:19:47.705989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.chdir('/kaggle/working/yolov5')\n\nbase_dir = 'base_dir'\nos.mkdir(base_dir)\n\n# images\nimages = os.path.join(base_dir, 'images')\nos.mkdir(images)\n\n# labels\nlabels = os.path.join(base_dir, 'labels')\nos.mkdir(labels)\n\n# create new folders inside images\ntrain = os.path.join(images, 'train')\nos.mkdir(train)\nvalidation = os.path.join(images, 'validation')\nos.mkdir(validation)\n\n\n# create new folders inside labels\ntrain = os.path.join(labels, 'train')\nos.mkdir(train)\nvalidation = os.path.join(labels, 'validation')\nos.mkdir(validation)\n\n# Display the folder structure\n!tree base_dir","metadata":{"papermill":{"duration":1.026715,"end_time":"2021-07-18T05:39:02.410223","exception":false,"start_time":"2021-07-18T05:39:01.383508","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:48.535764Z","iopub.execute_input":"2022-10-25T08:19:48.536143Z","iopub.status.idle":"2022-10-25T08:19:49.60612Z","shell.execute_reply.started":"2022-10-25T08:19:48.536115Z","shell.execute_reply":"2022-10-25T08:19:49.604921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Process the data\n\nHere we will write a function to process the training and validation data. \n\nWe need to create a separate txt file for each image that contains the details of all the bounding boxes on that image.  This function will also move the training and val data into the directory structure that we created above. We won't need to do any image resizing for Yolo. It will do that automatically during training.","metadata":{"papermill":{"duration":0.376731,"end_time":"2021-07-18T05:39:03.198014","exception":false,"start_time":"2021-07-18T05:39:02.821283","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Change the working directory\nos.chdir('/kaggle/working/')","metadata":{"papermill":{"duration":0.227543,"end_time":"2021-07-18T05:39:03.850619","exception":false,"start_time":"2021-07-18T05:39:03.623076","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:49.609515Z","iopub.execute_input":"2022-10-25T08:19:49.609884Z","iopub.status.idle":"2022-10-25T08:19:49.615248Z","shell.execute_reply.started":"2022-10-25T08:19:49.609852Z","shell.execute_reply":"2022-10-25T08:19:49.614045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Choose the fold to train on.\n\nfold_index = CHOSEN_FOLD\n\ndf_train = df_data[df_data['fold'] != fold_index]\ndf_val = df_data[df_data['fold'] == fold_index]\n\nprint(df_train['target'].value_counts())\nprint(df_val['target'].value_counts())","metadata":{"papermill":{"duration":0.259454,"end_time":"2021-07-18T05:39:04.321393","exception":false,"start_time":"2021-07-18T05:39:04.061939","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:50.064201Z","iopub.execute_input":"2022-10-25T08:19:50.064889Z","iopub.status.idle":"2022-10-25T08:19:50.080473Z","shell.execute_reply.started":"2022-10-25T08:19:50.064846Z","shell.execute_reply":"2022-10-25T08:19:50.079099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_data_for_yolo(df, data_type='train'):\n\n    for _, row in tq.tqdm(df.iterrows(), total=len(df)):\n        target = row['target']\n        study_slice = row['study_slice']\n        fname = study_slice + '.png'\n        if target == 1:\n            bbox_dict = row['boxes']\n            bbox_list = [bbox_dict]\n            image_width = row['w']\n            image_height = row['h']\n            \n            yolo_data = []\n            for coord_dict in bbox_list:\n\n                xmin = int(coord_dict['x'])\n                ymin = int(coord_dict['y'])\n                bbox_w = int(coord_dict['width'])\n                bbox_h = int(coord_dict['height'])\n                class_id = target\n\n                x_center = xmin + (bbox_w/2)\n                y_center = ymin + (bbox_h/2)\n\n                x_center = x_center/image_width\n                y_center = y_center/image_height\n                bbox_w = bbox_w/image_width\n                bbox_h = bbox_h/image_height\n\n                yolo_list = [class_id, x_center, y_center, bbox_w, bbox_h]\n\n                yolo_data.append(yolo_list)\n\n            yolo_data = np.array(yolo_data)\n            np.savetxt(os.path.join('yolov5/base_dir', \n                        f\"labels/{data_type}/{study_slice}.txt\"),\n                        yolo_data, \n                        fmt=[\"%d\", \"%f\", \"%f\", \"%f\", \"%f\"]\n                        )\n\n        shutil.copyfile(\n            f\"{prep_data_path}/images_dir/{fname}\",\n            os.path.join('yolov5/base_dir', f\"images/{data_type}/{fname}\")\n        )\nprocess_data_for_yolo(df_train, data_type='train')\nprocess_data_for_yolo(df_val, data_type='validation')","metadata":{"papermill":{"duration":3.555252,"end_time":"2021-07-18T05:39:11.818383","exception":false,"start_time":"2021-07-18T05:39:08.263131","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:19:50.737199Z","iopub.execute_input":"2022-10-25T08:19:50.73758Z","iopub.status.idle":"2022-10-25T08:20:49.003593Z","shell.execute_reply.started":"2022-10-25T08:19:50.737529Z","shell.execute_reply":"2022-10-25T08:20:49.002538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:20:49.005815Z","iopub.execute_input":"2022-10-25T08:20:49.006435Z","iopub.status.idle":"2022-10-25T08:20:49.944216Z","shell.execute_reply.started":"2022-10-25T08:20:49.006399Z","shell.execute_reply":"2022-10-25T08:20:49.943043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(os.listdir('yolov5/base_dir/images/train')))\nprint(len(os.listdir('yolov5/base_dir/images/validation')))\n\nprint(len(os.listdir('yolov5/base_dir/labels/train')))\nprint(len(os.listdir('yolov5/base_dir/labels/validation')))","metadata":{"papermill":{"duration":0.237175,"end_time":"2021-07-18T05:39:12.278949","exception":false,"start_time":"2021-07-18T05:39:12.041774","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:20:49.945858Z","iopub.execute_input":"2022-10-25T08:20:49.946168Z","iopub.status.idle":"2022-10-25T08:20:49.966105Z","shell.execute_reply.started":"2022-10-25T08:20:49.946138Z","shell.execute_reply":"2022-10-25T08:20:49.965061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_file_list = os.listdir('yolov5/base_dir/labels/train')\n\ntext_file = text_file_list[0]\n\ntext_file","metadata":{"papermill":{"duration":0.234228,"end_time":"2021-07-18T05:39:12.743254","exception":false,"start_time":"2021-07-18T05:39:12.509026","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:20:49.968974Z","iopub.execute_input":"2022-10-25T08:20:49.969344Z","iopub.status.idle":"2022-10-25T08:20:49.98136Z","shell.execute_reply.started":"2022-10-25T08:20:49.969306Z","shell.execute_reply":"2022-10-25T08:20:49.980493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! cat 'yolov5/base_dir/labels/train/1.2.826.0.1.3680043.26979_172.txt'","metadata":{"papermill":{"duration":1.008666,"end_time":"2021-07-18T05:39:13.974188","exception":false,"start_time":"2021-07-18T05:39:12.965522","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:20:49.984149Z","iopub.execute_input":"2022-10-25T08:20:49.984943Z","iopub.status.idle":"2022-10-25T08:20:50.922215Z","shell.execute_reply.started":"2022-10-25T08:20:49.98491Z","shell.execute_reply":"2022-10-25T08:20:50.921045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create the yaml file\nYolo requires that we also create a yaml file inside the yolov5 folder.","metadata":{"papermill":{"duration":0.222359,"end_time":"2021-07-18T05:39:14.419466","exception":false,"start_time":"2021-07-18T05:39:14.197107","status":"completed"},"tags":[]}},{"cell_type":"code","source":"yaml_dict = {'train': 'base_dir/images/train',   # path to the train folder\n            'val': 'base_dir/images/validation', # path to the val folder\n            'nc': 2,                             # number of classes\n            'names': ['0', '1']}                # list of label names\n\nimport yaml\n\nwith open(r'yolov5/my_data.yaml', 'w') as file:\n    documents = yaml.dump(yaml_dict, file)","metadata":{"papermill":{"duration":0.245553,"end_time":"2021-07-18T05:39:14.886293","exception":false,"start_time":"2021-07-18T05:39:14.64074","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:20:50.923946Z","iopub.execute_input":"2022-10-25T08:20:50.92425Z","iopub.status.idle":"2022-10-25T08:20:50.932806Z","shell.execute_reply.started":"2022-10-25T08:20:50.92422Z","shell.execute_reply":"2022-10-25T08:20:50.931623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('yolov5')","metadata":{"papermill":{"duration":0.233988,"end_time":"2021-07-18T05:39:15.342968","exception":false,"start_time":"2021-07-18T05:39:15.10898","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:20:50.934397Z","iopub.execute_input":"2022-10-25T08:20:50.935124Z","iopub.status.idle":"2022-10-25T08:20:50.944174Z","shell.execute_reply.started":"2022-10-25T08:20:50.935089Z","shell.execute_reply":"2022-10-25T08:20:50.94298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! cat 'yolov5/my_data.yaml'","metadata":{"papermill":{"duration":1.026811,"end_time":"2021-07-18T05:39:16.593576","exception":false,"start_time":"2021-07-18T05:39:15.566765","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:20:50.945736Z","iopub.execute_input":"2022-10-25T08:20:50.946219Z","iopub.status.idle":"2022-10-25T08:20:51.935263Z","shell.execute_reply.started":"2022-10-25T08:20:50.946185Z","shell.execute_reply":"2022-10-25T08:20:51.933651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create a custom hyperameter/augmentation yaml file","metadata":{"papermill":{"duration":0.223307,"end_time":"2021-07-18T05:39:17.039924","exception":false,"start_time":"2021-07-18T05:39:16.816617","status":"completed"},"tags":[]}},{"cell_type":"code","source":"yaml_dict = {\n    \n'lr0': 0.01,  # initial learning rate (SGD=1E-2, Adam=1E-3)\n'lrf': 0.032,  # final OneCycleLR learning rate (lr0 * lrf)\n'momentum': 0.937,  # SGD momentum/Adam beta1\n'weight_decay': 0.0005,  # optimizer weight decay 5e-4\n'warmup_epochs': 3.0,  # warmup epochs (fractions ok)\n'warmup_momentum': 0.8,  # warmup initial momentum\n'warmup_bias_lr': 0.1,  # warmup initial bias lr\n'box': 0.1,  # box loss gain\n'cls': 1.0,  # cls loss gain\n'cls_pw': 0.5,  # cls BCELoss positive_weight\n'obj': 2.0,  # obj loss gain (scale with pixels)\n'obj_pw': 0.5,  # obj BCELoss positive_weight\n'iou_t': 0.20,  # IoU training threshold\n'anchor_t': 4.0,  # anchor-multiple threshold\n'anchors': 0,  # anchors per output layer (0 to ignore)\n'fl_gamma': 0.0,  # focal loss gamma (efficientDet default gamma=1.5)\n'hsv_h': 0,  # image HSV-Hue augmentation (fraction)\n'hsv_s': 0,  # image HSV-Saturation augmentation (fraction)\n'hsv_v': 0,  # image HSV-Value augmentation (fraction)\n'degrees': 30.0,  # image rotation (+/- deg)\n'translate': 0.2,  # image translation (+/- fraction)\n'scale': 0.3,  # image scale (+/- gain)\n'shear': 0.0,  # image shear (+/- deg)\n'perspective': 0.0,  # image perspective (+/- fraction), range 0-0.001\n'flipud': 0.2,  # image flip up-down (probability)\n'fliplr': 0.5,  # image flip left-right (probability)\n'mosaic': 0.8,  # image mosaic (probability)\n'mixup': 0.0  # image mixup (probability)\n    \n}\nimport yaml\n\nwith open(r'yolov5/my_hyp.yaml', 'w') as file:\n    documents = yaml.dump(yaml_dict, file)","metadata":{"papermill":{"duration":0.239227,"end_time":"2021-07-18T05:39:17.506221","exception":false,"start_time":"2021-07-18T05:39:17.266994","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:20:51.937697Z","iopub.execute_input":"2022-10-25T08:20:51.938801Z","iopub.status.idle":"2022-10-25T08:20:51.967492Z","shell.execute_reply.started":"2022-10-25T08:20:51.938749Z","shell.execute_reply":"2022-10-25T08:20:51.966396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('yolov5')","metadata":{"papermill":{"duration":0.234891,"end_time":"2021-07-18T05:39:17.96441","exception":false,"start_time":"2021-07-18T05:39:17.729519","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:20:51.975447Z","iopub.execute_input":"2022-10-25T08:20:51.975733Z","iopub.status.idle":"2022-10-25T08:20:51.991388Z","shell.execute_reply.started":"2022-10-25T08:20:51.975707Z","shell.execute_reply":"2022-10-25T08:20:51.987313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! cat 'yolov5/my_hyp.yaml'","metadata":{"papermill":{"duration":1.010899,"end_time":"2021-07-18T05:39:19.194646","exception":false,"start_time":"2021-07-18T05:39:18.183747","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:20:51.996778Z","iopub.execute_input":"2022-10-25T08:20:51.997227Z","iopub.status.idle":"2022-10-25T08:20:52.984513Z","shell.execute_reply.started":"2022-10-25T08:20:51.997181Z","shell.execute_reply":"2022-10-25T08:20:52.98332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train the model","metadata":{"papermill":{"duration":0.239099,"end_time":"2021-07-18T05:39:19.65831","exception":false,"start_time":"2021-07-18T05:39:19.419211","status":"completed"},"tags":[]}},{"cell_type":"code","source":"os.chdir('/kaggle/working/yolov5')\n!pwd","metadata":{"papermill":{"duration":1.119913,"end_time":"2021-07-18T05:39:21.005032","exception":false,"start_time":"2021-07-18T05:39:19.885119","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:20:52.98624Z","iopub.execute_input":"2022-10-25T08:20:52.986644Z","iopub.status.idle":"2022-10-25T08:20:53.925054Z","shell.execute_reply.started":"2022-10-25T08:20:52.986611Z","shell.execute_reply":"2022-10-25T08:20:53.9239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# What some of the parameters mean:\n\n# --weights => the pre-trained model that we are using.\n# We are using weights that have been downloaded and stored in a Kaggle dataset.\n\n# --save-txt => The predicted bbox coordinates get saved to a txt file. One txt file per image.\n# --save-conf => The conf score gets included in the above txt file.\n# --img => The image will be resized to this size before creating the mosaic.\n# --conf => The confidence threshold\n# --rect => Means don't use mosaic augmentation during training\n# --name => Give a model a name e.g. --name my_model\n# --batch => batch size\n# --epochs => number of training epochs\n# --data => the yaml file path\n# --exist-ok => do not increment the project names with each run i.e. don't change exp to epx2, exp3 etc.\n# --nosave => do not save the images/videos (helpful when deploying to a server)\n\n# It's helpful to review the source code in detect.py to know what the above parameters mean.\n# detect.py is located inside the yolov5 folder.","metadata":{"papermill":{"duration":0.234227,"end_time":"2021-07-18T05:39:21.512445","exception":false,"start_time":"2021-07-18T05:39:21.278218","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:08:15.751923Z","iopub.execute_input":"2022-10-25T08:08:15.75226Z","iopub.status.idle":"2022-10-25T08:08:15.758136Z","shell.execute_reply.started":"2022-10-25T08:08:15.752226Z","shell.execute_reply":"2022-10-25T08:08:15.757066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yolo_model_path = '/kaggle/input/yolov54/yolov5l.pt'\n!WANDB_MODE=\"dryrun\" python train.py --img $IMAGE_SIZE --batch $BATCH_SIZE --epochs $NUM_EPOCHS --data my_data.yaml --hyp my_hyp.yaml --weights $yolo_model_path","metadata":{"_kg_hide-output":true,"papermill":{"duration":4424.742752,"end_time":"2021-07-18T06:53:06.478556","exception":false,"start_time":"2021-07-18T05:39:21.735804","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:05:35.134159Z","iopub.status.idle":"2022-10-25T08:05:35.134955Z","shell.execute_reply.started":"2022-10-25T08:05:35.134672Z","shell.execute_reply":"2022-10-25T08:05:35.134698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Copy the trained model\n\nWe will copy the trained model to the Kaggle working directory. This will make the model easier to access.","metadata":{"papermill":{"duration":3.382839,"end_time":"2021-07-18T06:53:20.977817","exception":false,"start_time":"2021-07-18T06:53:17.594978","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# This is where the trained model is stored\n\npath = 'runs/train/exp/weights/'\n\nos.listdir(path)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# change the working directory\nos.chdir('/kaggle/working/')\n\n!pwd","metadata":{"papermill":{"duration":4.147132,"end_time":"2021-07-18T06:53:35.505998","exception":false,"start_time":"2021-07-18T06:53:31.358866","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:08:26.519011Z","iopub.execute_input":"2022-10-25T08:08:26.519395Z","iopub.status.idle":"2022-10-25T08:08:27.477895Z","shell.execute_reply.started":"2022-10-25T08:08:26.51934Z","shell.execute_reply":"2022-10-25T08:08:27.476726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Copy the best model to the kaggle/working\n\nshutil.copyfile(\n    '/kaggle/working/yolov5/runs/train/exp/weights/best.pt',\n    '/kaggle/working/best.pt')\n\n!ls","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get the name of the last experiment\n\nYolov5 saves every training run as an experiment.","metadata":{"papermill":{"duration":3.532048,"end_time":"2021-07-18T06:53:51.514652","exception":false,"start_time":"2021-07-18T06:53:47.982604","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# change the working directory to yolov5\nos.chdir('/kaggle/working/yolov5')\n!pwd","metadata":{"papermill":{"duration":4.55418,"end_time":"2021-07-18T06:53:59.475607","exception":false,"start_time":"2021-07-18T06:53:54.921427","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:08:31.687885Z","iopub.execute_input":"2022-10-25T08:08:31.688578Z","iopub.status.idle":"2022-10-25T08:08:32.788282Z","shell.execute_reply.started":"2022-10-25T08:08:31.688538Z","shell.execute_reply":"2022-10-25T08:08:32.786921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('runs/train/')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get a list of experiments\nexp_list = os.listdir('runs/train/')\n\nexp_list","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"exp = exp_list[0]\nexp","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the contents of the \"exp\" folder\nos.listdir(f'runs/train/{exp}')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"| <a id='validation_review'></a>","metadata":{}},{"cell_type":"markdown","source":"## Training and Validation Review\n\n**Please look at the list of files in the output of the above cell.**\n\n- Yolo stores all the training curves as one png file. To view the training curves we need to display the png file.\n\n- The summary displayed at the end of training shows the resuts for the LAST epoch. This is not the results for the BEST epoch. The results for each epoch are logged in a file called results.txt. We will load that file into a pandas dataframe and then get the best epoch and the best mAP score.\n\n- Yolo also stores png images showing the true and predicted labels for each val batch. In these images the true and predicted bounding boxes are drawn in. One batch is shown on one image.\n\nThere's more results info available. The yolov5 folder will appear in the output of this notebook. I suggest that you download it and look at the contents of the yolov5/runs/train/exp folder.","metadata":{"papermill":{"duration":3.34843,"end_time":"2021-07-18T06:54:34.867454","exception":false,"start_time":"2021-07-18T06:54:31.519024","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Display the contents of the \"exp\" folder\nos.listdir(f'runs/train/{exp}')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.chdir('/kaggle/working/yolov5')\n!pwd","metadata":{"papermill":{"duration":4.17146,"end_time":"2021-07-18T06:54:49.377868","exception":false,"start_time":"2021-07-18T06:54:45.206408","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:20:53.927134Z","iopub.execute_input":"2022-10-25T08:20:53.927552Z","iopub.status.idle":"2022-10-25T08:20:54.88969Z","shell.execute_reply.started":"2022-10-25T08:20:53.927513Z","shell.execute_reply":"2022-10-25T08:20:54.888316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Display the training curves","metadata":{"papermill":{"duration":3.348482,"end_time":"2021-07-18T06:54:56.99484","exception":false,"start_time":"2021-07-18T06:54:53.646358","status":"completed"},"tags":[]}},{"cell_type":"code","source":"plt.figure(figsize = (15, 15))\nplt.imshow(plt.imread(f'runs/train/{exp}/results.png'))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get the best mAP and best epoch","metadata":{"papermill":{"duration":3.325504,"end_time":"2021-07-18T06:55:11.376077","exception":false,"start_time":"2021-07-18T06:55:08.050573","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!ls","metadata":{"papermill":{"duration":4.486321,"end_time":"2021-07-18T06:55:19.433214","exception":false,"start_time":"2021-07-18T06:55:14.946893","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:08:46.53569Z","iopub.execute_input":"2022-10-25T08:08:46.536077Z","iopub.status.idle":"2022-10-25T08:08:47.587682Z","shell.execute_reply.started":"2022-10-25T08:08:46.536044Z","shell.execute_reply":"2022-10-25T08:08:47.585802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = f'runs/train/{exp}/results.txt'\n!cat $path","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filename = f'runs/train/{exp}/results.txt'\nfile_list = []\nwith open(filename) as f:\n    file_line_list = f.readlines()\n    \n    \nfor i in range(0, len(file_line_list)):\n    line_list = file_line_list[i].split()\n    line_list = [x.strip() for x in line_list]\n    file_list.append(line_list)\n    \nlen(file_list)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame(file_list)\ndf.head(10)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col_names = ['epoch', 'P', 'R', 'map0.5', 'map0.5:0.95']\ndf_results = df[[0, 8, 9, 10, 11]]\ndf_results.columns = col_names\ndf_results.head(10)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_map = df_results['map0.5'].max()\n\nprint('---------------------')\n\nprint('Best map0.5:', best_map)\nprint()\n\ndf = df_results[df_results['map0.5'] == best_map]\n\nprint(df.head())\n\nprint('---------------------')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Display one batch of train images","metadata":{"execution":{"iopub.execute_input":"2021-07-05T08:19:53.683161Z","iopub.status.busy":"2021-07-05T08:19:53.682523Z","iopub.status.idle":"2021-07-05T08:19:53.68887Z","shell.execute_reply":"2021-07-05T08:19:53.687579Z","shell.execute_reply.started":"2021-07-05T08:19:53.683105Z"},"papermill":{"duration":3.572099,"end_time":"2021-07-18T06:56:03.775207","exception":false,"start_time":"2021-07-18T06:56:00.203108","status":"completed"},"tags":[]}},{"cell_type":"code","source":"plt.figure(figsize = (15, 15))\nplt.imshow(plt.imread(f'runs/train/{exp}/train_batch0.jpg'))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Display true and predicted val set bboxes\n\nHere we will display the true and predicted bboxes for two val batches.","metadata":{"papermill":{"duration":3.378424,"end_time":"2021-07-18T06:56:18.287695","exception":false,"start_time":"2021-07-18T06:56:14.909271","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# BATCH 0 - TRUE BBOXES\n\nplt.figure(figsize = (15, 15))\nplt.imshow(plt.imread(f'runs/train/{exp}/test_batch0_labels.jpg'))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# BATCH 0 - PREDICTED BBOXES\n\nplt.figure(figsize = (15, 15))\nplt.imshow(plt.imread(f'runs/train/{exp}/test_batch0_pred.jpg'))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# BATCH 1 - TRUE BBOXES\n\nplt.figure(figsize = (15, 15))\nplt.imshow(plt.imread(f'runs/train/{exp}/test_batch1_labels.jpg'))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# BATCH 1 - PREDICTED BBOXES\n\nplt.figure(figsize = (15, 15))\nplt.imshow(plt.imread(f'runs/train/{exp}/test_batch1_pred.jpg'))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Make a prediction on the val set","metadata":{"papermill":{"duration":4.04488,"end_time":"2021-07-18T06:57:04.446831","exception":false,"start_time":"2021-07-18T06:57:00.401951","status":"completed"},"tags":[]}},{"cell_type":"code","source":"os.chdir('/kaggle/working')\n!pwd","metadata":{"papermill":{"duration":4.204321,"end_time":"2021-07-18T06:57:12.466427","exception":false,"start_time":"2021-07-18T06:57:08.262106","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:21:12.329214Z","iopub.execute_input":"2022-10-25T08:21:12.329635Z","iopub.status.idle":"2022-10-25T08:21:13.277022Z","shell.execute_reply.started":"2022-10-25T08:21:12.329601Z","shell.execute_reply":"2022-10-25T08:21:13.275714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a folder to store the extracted files\nif os.path.isdir('yolo_images_dir') == False:\n    yolo_images_dir = 'yolo_images_dir'\n    os.mkdir(yolo_images_dir)\n    \n    \n    \nval_fname_list = list(df_val['study_slice'])\n\nfor study_slice in val_fname_list:\n    \n    fname = study_slice + '.png'\n    \n    shutil.copyfile(\n        f\"{prep_data_path}/images_dir/{fname}\",\n        f\"yolo_images_dir/{fname}\")\n    \n    \nlen(os.listdir('yolo_images_dir'))","metadata":{"papermill":{"duration":5.227336,"end_time":"2021-07-18T06:57:28.383883","exception":false,"start_time":"2021-07-18T06:57:23.156547","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:21:16.067135Z","iopub.execute_input":"2022-10-25T08:21:16.067525Z","iopub.status.idle":"2022-10-25T08:21:18.130488Z","shell.execute_reply.started":"2022-10-25T08:21:16.067493Z","shell.execute_reply":"2022-10-25T08:21:18.129465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"papermill":{"duration":4.662857,"end_time":"2021-07-18T06:57:36.513146","exception":false,"start_time":"2021-07-18T06:57:31.850289","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:21:23.612382Z","iopub.execute_input":"2022-10-25T08:21:23.61276Z","iopub.status.idle":"2022-10-25T08:21:24.551172Z","shell.execute_reply.started":"2022-10-25T08:21:23.612728Z","shell.execute_reply":"2022-10-25T08:21:24.550002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# change the working directory\nos.chdir('/kaggle/working/yolov5')\n\n!pwd","metadata":{"papermill":{"duration":4.277636,"end_time":"2021-07-18T06:57:44.58411","exception":false,"start_time":"2021-07-18T06:57:40.306474","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:21:24.566524Z","iopub.execute_input":"2022-10-25T08:21:24.56689Z","iopub.status.idle":"2022-10-25T08:21:25.502924Z","shell.execute_reply.started":"2022-10-25T08:21:24.566859Z","shell.execute_reply":"2022-10-25T08:21:25.501753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make a prediction on all images in images_dir\n# The model only creates a txt file if it finds objects on an image.\n\ntest_images_path = '/kaggle/working/yolo_images_dir'\nyolo_model_path = '/kaggle/working/best.pt'\n\n!python detect.py --source $test_images_path --weights $yolo_model_path --img $IMAGE_SIZE --save-txt --save-conf --exist-ok","metadata":{"_kg_hide-output":true,"papermill":{"duration":41.782806,"end_time":"2021-07-18T06:58:29.775248","exception":false,"start_time":"2021-07-18T06:57:47.992442","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:03:10.177568Z","iopub.execute_input":"2022-10-25T08:03:10.178015Z","iopub.status.idle":"2022-10-25T08:04:24.203026Z","shell.execute_reply.started":"2022-10-25T08:03:10.177973Z","shell.execute_reply":"2022-10-25T08:04:24.201791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:22:06.01898Z","iopub.execute_input":"2022-10-25T08:22:06.019404Z","iopub.status.idle":"2022-10-25T08:22:06.025113Z","shell.execute_reply.started":"2022-10-25T08:22:06.019367Z","shell.execute_reply":"2022-10-25T08:22:06.024113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds = pydicom.dcmread('/kaggle/working/100.dcm')\nnew_image = ds.pixel_array.astype(float)","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:22:06.756941Z","iopub.execute_input":"2022-10-25T08:22:06.757309Z","iopub.status.idle":"2022-10-25T08:22:06.769702Z","shell.execute_reply.started":"2022-10-25T08:22:06.75728Z","shell.execute_reply":"2022-10-25T08:22:06.768624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaled_image = (np.maximum(new_image, 0) / new_image.max()) * 255.0","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:22:07.998373Z","iopub.execute_input":"2022-10-25T08:22:07.998761Z","iopub.status.idle":"2022-10-25T08:22:08.007239Z","shell.execute_reply.started":"2022-10-25T08:22:07.998724Z","shell.execute_reply":"2022-10-25T08:22:08.006255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaled_image = np.uint8(scaled_image)\nfinal_image = Image.fromarray(scaled_image)","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:22:08.836666Z","iopub.execute_input":"2022-10-25T08:22:08.837352Z","iopub.status.idle":"2022-10-25T08:22:08.843227Z","shell.execute_reply.started":"2022-10-25T08:22:08.837319Z","shell.execute_reply":"2022-10-25T08:22:08.842109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_image.save('/kaggle/working/image.png')","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:22:09.328678Z","iopub.execute_input":"2022-10-25T08:22:09.329031Z","iopub.status.idle":"2022-10-25T08:22:09.383674Z","shell.execute_reply.started":"2022-10-25T08:22:09.329003Z","shell.execute_reply":"2022-10-25T08:22:09.382768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pwd","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:22:10.625889Z","iopub.execute_input":"2022-10-25T08:22:10.626951Z","iopub.status.idle":"2022-10-25T08:22:10.63483Z","shell.execute_reply.started":"2022-10-25T08:22:10.626905Z","shell.execute_reply":"2022-10-25T08:22:10.633619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Tamoghno**","metadata":{}},{"cell_type":"code","source":"# Make a prediction on all images in images_dir\n\n# The model only creates a txt file if it finds objects on an image.\n\ntest_image_path = '/kaggle/working/image.png'\nyolo_model_path = '/kaggle/working/best.pt'\n\n# Ensembling two Yolo models\n# How to ensemble Yolov5 models:\n# Ref: https://github.com/ultralytics/yolov5/issues/318\n\n!python detect.py --source $test_image_path --weights $yolo_model_path --img $IMAGE_SIZE --save-txt --save-conf --exist-ok","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:22:27.343236Z","iopub.execute_input":"2022-10-25T08:22:27.343638Z","iopub.status.idle":"2022-10-25T08:22:35.232863Z","shell.execute_reply.started":"2022-10-25T08:22:27.343607Z","shell.execute_reply":"2022-10-25T08:22:35.231582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Process the predictions","metadata":{"papermill":{"duration":3.642808,"end_time":"2021-07-18T06:58:37.370203","exception":false,"start_time":"2021-07-18T06:58:33.727395","status":"completed"},"tags":[]}},{"cell_type":"code","source":"ls","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:22:43.673613Z","iopub.execute_input":"2022-10-25T08:22:43.674003Z","iopub.status.idle":"2022-10-25T08:22:44.620133Z","shell.execute_reply.started":"2022-10-25T08:22:43.67397Z","shell.execute_reply":"2022-10-25T08:22:44.618943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"txt_files_list = os.listdir('runs/detect/exp/labels')\n\nprint(len(txt_files_list))\nprint(txt_files_list[0])","metadata":{"papermill":{"duration":4.536142,"end_time":"2021-07-18T06:58:45.507384","exception":false,"start_time":"2021-07-18T06:58:40.971242","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-25T08:22:44.935961Z","iopub.execute_input":"2022-10-25T08:22:44.936348Z","iopub.status.idle":"2022-10-25T08:22:44.944643Z","shell.execute_reply.started":"2022-10-25T08:22:44.936315Z","shell.execute_reply":"2022-10-25T08:22:44.943416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Put the info inside all the txt files into one dataframe.\n# Remember that if the image does not have any bounding boxes\n# then Yolo does not create a txt file for it.\n\ntxt_files_list = os.listdir('runs/detect/exp/labels')\n\nfor i, txt_file in enumerate(txt_files_list):\n    path = f'runs/detect/exp/labels/{txt_file}'\n    cols = ['class', 'x-center', 'y-center', 'bbox_width', 'bbox_height', 'conf-score']\n    df = pd.read_csv(path, sep=\" \", header=None)\n    df.columns = cols\n    fname = txt_file.replace(\"txt\", \"png\")\n    df['id'] = fname\n    if i == 0:\n        df_test_preds = df\n    else:\n        df_test_preds = pd.concat([df_test_preds, df], axis=0)\n       \n    \n    \nprint(len(txt_files_list))\nprint(df_test_preds['id'].nunique())\nprint(df_test_preds.shape)\n\ndf_test_preds.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-25T08:22:48.212665Z","iopub.execute_input":"2022-10-25T08:22:48.21336Z","iopub.status.idle":"2022-10-25T08:22:48.238551Z","shell.execute_reply.started":"2022-10-25T08:22:48.213325Z","shell.execute_reply":"2022-10-25T08:22:48.237587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imread('')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_val = df_val.reset_index(drop=True)\n\ndf_val['id'] =  df_val['study_slice'] + '.png'\nval_pred_list = []\npred_list = list(df_test_preds['id'])\n\nfor i in range(0, len(df_val)):\n    fname = df_val.loc[i, 'id']\n    if fname in pred_list:\n        val_pred_list.append(1)\n    else:\n        val_pred_list.append(0)\n    \n    \ndf_val['preds'] = val_pred_list\ndf_val['preds'].value_counts()","metadata":{"papermill":{"duration":3.732707,"end_time":"2021-07-18T06:59:02.462832","exception":false,"start_time":"2021-07-18T06:58:58.730125","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-29T08:33:36.590016Z","iopub.execute_input":"2022-09-29T08:33:36.590488Z","iopub.status.idle":"2022-09-29T08:33:36.67488Z","shell.execute_reply.started":"2022-09-29T08:33:36.590447Z","shell.execute_reply":"2022-09-29T08:33:36.673794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Confusion Matrix","metadata":{"papermill":{"duration":3.706424,"end_time":"2021-07-18T06:59:10.094022","exception":false,"start_time":"2021-07-18T06:59:06.387598","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n\nCLASS_LIST = ['Normal', 'Fracture']\n    \n# targets\ny_true = list(df_val['target'])\n\n# get the preds as integers\ny_pred = list(df_val['preds'])\n\n# argmax returns the index of the max value in each row.\ncm = confusion_matrix(y_true, y_pred)\n\n# Display the confusion matrix.\nprint()\nprint(cm)\nprint(CLASS_LIST)","metadata":{"papermill":{"duration":4.773196,"end_time":"2021-07-18T06:59:18.504453","exception":false,"start_time":"2021-07-18T06:59:13.731257","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-29T08:33:36.676519Z","iopub.execute_input":"2022-09-29T08:33:36.6769Z","iopub.status.idle":"2022-09-29T08:33:36.688043Z","shell.execute_reply.started":"2022-09-29T08:33:36.676855Z","shell.execute_reply":"2022-09-29T08:33:36.686934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm_plot_labels = ['normal', 'fracture']\ntext_size=12\nplot_confusion_matrix(cm, cm_plot_labels, title='Confusion Matrix', text_size=text_size)","metadata":{"papermill":{"duration":3.838768,"end_time":"2021-07-18T06:59:26.204345","exception":false,"start_time":"2021-07-18T06:59:22.365577","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-29T08:33:39.964038Z","iopub.execute_input":"2022-09-29T08:33:39.964459Z","iopub.status.idle":"2022-09-29T08:33:40.203106Z","shell.execute_reply.started":"2022-09-29T08:33:39.964427Z","shell.execute_reply":"2022-09-29T08:33:40.202182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Classification Report","metadata":{"papermill":{"duration":3.649494,"end_time":"2021-07-18T06:59:33.878606","exception":false,"start_time":"2021-07-18T06:59:30.229112","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n    \nreport = classification_report(y_true, y_pred, target_names=cm_plot_labels)\n\nprint()\nprint(report)","metadata":{"papermill":{"duration":4.01076,"end_time":"2021-07-18T06:59:41.630028","exception":false,"start_time":"2021-07-18T06:59:37.619268","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-29T08:33:44.380123Z","iopub.execute_input":"2022-09-29T08:33:44.380803Z","iopub.status.idle":"2022-09-29T08:33:44.395821Z","shell.execute_reply.started":"2022-09-29T08:33:44.380767Z","shell.execute_reply":"2022-09-29T08:33:44.394942Z"},"trusted":true},"execution_count":null,"outputs":[]}]}