{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold \nfrom tqdm import tqdm\nimport shutil \nimport yaml\nimport torch\nfrom IPython.display import Image, clear_output\nfrom glob import glob","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df = pd.read_csv('../input/custom-csv/test_meta.csv')\ntest_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''\ntrain_df = pd.read_csv('../input/vinbigdata-chest-xray-abnormalities-detection/train.csv')\ntrain_meta = pd.read_csv('../input/custom-csv/train_meta.csv')\n\ndf = pd.merge(train_df, train_meta, on='image_id')\ndf.head()\n'''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''\nCFG = {\n    'dim': 512,\n    'fold': 4,\n}\n\ndf.fillna(0, inplace=True)\n\n#Time to fix the other two coordinate values of the \"No finding\" category\ndf.loc[df[\"class_id\"] == 14, [\"x_max\", \"y_max\"]] = 1.0\n\ndf['x_min'] = df.apply(lambda row: (row.x_min)/row.dim1, axis =1)\n\ndf['y_min'] = df.apply(lambda row: (row.y_min)/row.dim0, axis =1)\n\ndf['x_max'] = df.apply(lambda row: (row.x_max)/row.dim1, axis =1)\n\ndf['y_max'] = df.apply(lambda row: (row.y_max)/row.dim0, axis =1)\n\ndf['x_mid'] = df.apply(lambda row: (row.x_max+row.x_min)/2, axis =1)\n\ndf['y_mid'] = df.apply(lambda row: (row.y_max+row.y_min)/2, axis =1)\n\ndf['w'] = df.apply(lambda row: (row.x_max-row.x_min), axis =1)\n\ndf['h'] = df.apply(lambda row: (row.y_max-row.y_min), axis =1)\n\ndf['area'] = df['w']*df['h']\n\ndf.head()\n\nfeatures = ['x_min', 'y_min', 'x_max', 'y_max', 'x_mid', 'y_mid', 'w', 'h', 'area']\nX = df[features]\ny = df['class_id']\nX.shape, y.shape\n\n\nskf = StratifiedKFold(n_splits=10, shuffle=True, random_state=123) \n\ndf['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(skf.split(X,y, groups = df.image_id.tolist())):\n    df.loc[val_idx, 'fold'] = fold\ndf.head()\n\ntrain_files = []\nval_files   = []\nval_files += list(df[df.fold==CFG['fold']].image_id.unique())\ntrain_files += list(df[df.fold!=CFG['fold']].image_id.unique())\nlen(train_files), len(val_files)\n\nos.makedirs('./vinbigdata-chest-xray-abnormalities-detection/train/labels', exist_ok = True)\nos.makedirs('./vinbigdata-chest-xray-abnormalities-detection/train/images', exist_ok = True)\nos.makedirs('./vinbigdata-chest-xray-abnormalities-detection/valid/labels', exist_ok = True)\nos.makedirs('./vinbigdata-chest-xray-abnormalities-detection/valid/images', exist_ok = True)\n\nTRAIN_LABELS_PATH = './vinbigdata-chest-xray-abnormalities-detection/train/labels'\nVAL_LABELS_PATH = './vinbigdata-chest-xray-abnormalities-detection/valid/labels'\n\nfor file in tqdm(train_files):\n    records = df[df['image_id'] == file]\n    attributes = records[['class_id','x_mid','y_mid','w','h']].values\n    attributes = np.array(attributes)\n    np.savetxt(\n        os.path.join(\n            TRAIN_LABELS_PATH,\n            f\"{file}.txt\"\n        ),\n        attributes,\n        fmt = [\"%d\",\"%f\",\"%f\",\"%f\",\"%f\"]\n    )\n    shutil.copy(f'../input/vinbigdata-jpg/train_jpg/{file}.jpg', './vinbigdata-chest-xray-abnormalities-detection/train/images')\n    \n    \n    \nfor file in tqdm(val_files):\n    records = df[df['image_id'] == file]\n    attributes = records[['class_id','x_mid','y_mid','w','h']].values\n    attributes = np.array(attributes)\n    np.savetxt(\n        os.path.join(\n            VAL_LABELS_PATH,\n            f\"{file}.txt\"\n        ),\n        attributes,\n        fmt = [\"%d\",\"%f\",\"%f\",\"%f\",\"%f\"]\n    )\n    shutil.copy(f'../input/vinbigdata-jpg/train_jpg/{file}.jpg', './vinbigdata-chest-xray-abnormalities-detection/train/images')\n    \n    \n    \nclass_ids, class_names = list(zip(*set(zip(df.class_id, df.class_name))))\nclasses = list(np.array(class_names)[np.argsort(class_ids)])\nclasses = list(map(lambda x: str(x), classes))\nclasses\n\n\ndata = dict(\n    train = '../vinbigdata-chest-xray-abnormalities-detection/train/images',\n    val   = '../vinbigdata-chest-xray-abnormalities-detection/valid/images',\n    nc    = 15,\n    names = classes\n    )\n\nwith open('./data.yaml', 'w') as outfile:\n    yaml.dump(data, outfile, default_flow_style=False)\n    \nf = open('data.yaml', 'r')\nprint('\\nyaml:')\nprint(f.read())\n\n'''\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"shutil.copytree('../input/yolov5-official-v31-dataset/yolov5', './yolov5')\nshutil.copytree('../input/yolov5-model/exp34/exp34', './yolov5/runs/train/exp34')\nshutil.copytree('../input/vinbigdata-jpg/test_jpg', './yolov5/data/test_data')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"os.chdir('./yolov5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!WANDB_MODE=\"dryrun\" python train.py --img 1024 --batch 16 --epochs 10 --data ../data.yaml --weights yolov5x.pt --cache","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!python detect.py --weights 'runs/train/exp34/weights/best.pt' --img 1024 --conf 0.15 --iou 0.5 --source data/test_data --exist-ok --save-txt --save-con","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def yolo2voc(image_height, image_width, bboxes):\n    \"\"\"\n    yolo => [xmid, ymid, w, h] (normalized)\n    voc  => [x1, y1, x2, y1]\n    \n    \"\"\" \n    bboxes = bboxes.copy().astype(float) # otherwise all value will be 0 as voc_pascal dtype is np.int\n    \n    bboxes[..., [0, 2]] = bboxes[..., [0, 2]]* image_width\n    bboxes[..., [1, 3]] = bboxes[..., [1, 3]]* image_height\n    \n    bboxes[..., [0, 1]] = bboxes[..., [0, 1]] - bboxes[..., [2, 3]]/2\n    bboxes[..., [2, 3]] = bboxes[..., [0, 1]] + bboxes[..., [2, 3]]\n    \n    return bboxes\n\nimage_ids = []\nPredictionStrings = []\n\nfor file_path in tqdm(glob('runs/detect/exp/labels/*txt')):\n    image_id = file_path.split('/')[-1].split('.')[0]\n    w, h = test_df.loc[test_df.image_id==image_id,['dim0', 'dim1']].values[0]\n    f = open(file_path, 'r')\n    data = np.array(f.read().replace('\\n', ' ').strip().split(' ')).astype(np.float32).reshape(-1, 6)\n    data = data[:, [0, 5, 1, 2, 3, 4]]\n    bboxes = list(np.round(np.concatenate((data[:, :2], np.round(yolo2voc(h, w, data[:, 2:]))), axis =1).reshape(-1), 1).astype(str))\n    for idx in range(len(bboxes)):\n        bboxes[idx] = str(int(float(bboxes[idx]))) if idx%6!=1 else bboxes[idx]\n    image_ids.append(image_id)\n    PredictionStrings.append(' '.join(bboxes))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_df = pd.DataFrame({'image_id':image_ids,\n                        'PredictionString':PredictionStrings})\nsub_df = pd.merge(test_df, pred_df, on = 'image_id', how = 'left').fillna(\"14 1 0 0 1 1\")\nsub_df = sub_df[['image_id', 'PredictionString']]\nsub_df.to_csv('/kaggle/working/submission.csv',index = False)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"shutil.rmtree('/kaggle/working/yolov5')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}