{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"! pip install pylibjpeg -q\n! pip install python-gdcm -q\n! pip install pylibjpeg-libjpeg -q","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:51:56.310032Z","iopub.execute_input":"2022-09-26T23:51:56.310478Z","iopub.status.idle":"2022-09-26T23:52:36.160955Z","shell.execute_reply.started":"2022-09-26T23:51:56.310387Z","shell.execute_reply":"2022-09-26T23:52:36.15949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from fastai.vision.all import *\nfrom fastai.medical.imaging import *\nfrom fastcore.all import *\n\nimport pandas as pd\nimport numpy as np\nimport pydicom\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport cv2\n\nfrom skimage.transform import resize","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:36.163497Z","iopub.execute_input":"2022-09-26T23:52:36.16403Z","iopub.status.idle":"2022-09-26T23:52:40.657635Z","shell.execute_reply.started":"2022-09-26T23:52:36.163987Z","shell.execute_reply":"2022-09-26T23:52:40.656297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Use the predictions to extract only cspine images\n\n8 sagittal slices were used for each exam because 16 caused memory issues. \nThe prediction masks need to be transformed to match the images and can then be used to get the cspine vs non cspine regions.\n\nThe next step is to test to see if we can get away with even less sagittal slices which will decrease prediction time on the test set. We will experiment with 4 and compare.\n\nResult: 4 works just as well as 8 with half the predictions needed.","metadata":{}},{"cell_type":"code","source":"root = Path('../input/rsna-2022-cervical-spine-fracture-detection/train_images')\njpg_root = Path('../input/segment-cspine-notebook-2-2')\ncsv_root = Path('../input/cspine-csvs')\n\nseg_path = Path('../input/rsna-2022-cervical-spine-fracture-detection/segmentations')\nsegmentations = list(seg_path.iterdir())\nseg_ids = [o.stem for o in segmentations]\n\nexclude = '1.2.826.0.1.3680043.20574'","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:40.659609Z","iopub.execute_input":"2022-09-26T23:52:40.66038Z","iopub.status.idle":"2022-09-26T23:52:40.689701Z","shell.execute_reply.started":"2022-09-26T23:52:40.66033Z","shell.execute_reply":"2022-09-26T23:52:40.688422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load csvs\n\n'Study ranges with cspine' has the range predictions from a model that classifies axial slices as cspine or not\n\n'Train dcm filenames' has the study id and slice number of every file in the train_images folder","metadata":{}},{"cell_type":"code","source":"rdf = pd.read_csv('../input/cspine-csvs/study_ranges_with_cspine.csv')\ndf = pd.read_csv(csv_root/'train_dcm_filenames.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:40.693175Z","iopub.execute_input":"2022-09-26T23:52:40.693537Z","iopub.status.idle":"2022-09-26T23:52:41.221745Z","shell.execute_reply.started":"2022-09-26T23:52:40.693504Z","shell.execute_reply":"2022-09-26T23:52:41.220568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get the count for each study id","metadata":{}},{"cell_type":"code","source":"count_df = df.groupby('study_id').count().reset_index()\ncount_df.rename(columns = {'slice':'n'}, inplace=True)\n#remove excluded case\ncount_df = count_df[count_df.study_id != exclude]\n#study_ids\nstudy_ids = count_df.study_id.to_list()\ncount_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:41.223095Z","iopub.execute_input":"2022-09-26T23:52:41.223435Z","iopub.status.idle":"2022-09-26T23:52:41.342669Z","shell.execute_reply.started":"2022-09-26T23:52:41.223406Z","shell.execute_reply":"2022-09-26T23:52:41.341529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utility functions","metadata":{}},{"cell_type":"code","source":"def get_fn(study_id, n, ext = '.dcm'):\n    if ext == '.dcm':\n        return root/study_id/f'{n}{ext}'\n    else:\n        return jpg_root/(study_id + f'_{n}{ext}')\n    \ndef show_case(study_id, aspect = 2):\n    fig, axs = plt.subplots(4,4, figsize=(16, 12))\n    axs = axs.flatten()\n    for i in range(0, len(axs), 2):\n        fn = get_fn(study_id, int(i/2), '.jpg')\n        img,mask = image_and_mask(fn)\n        axs[i].imshow(img, cmap='bone')\n        axs[i+1].imshow(mask)\n     \n        axs[i].axis(\"off\")\n        axs[i].set_aspect(aspect)\n        axs[i+1].axis(\"off\")\n        axs[i+1].set_aspect(aspect)\n        \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:41.343957Z","iopub.execute_input":"2022-09-26T23:52:41.344317Z","iopub.status.idle":"2022-09-26T23:52:41.354585Z","shell.execute_reply.started":"2022-09-26T23:52:41.344284Z","shell.execute_reply":"2022-09-26T23:52:41.353309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fn = Path(jpg_root/'1.2.826.0.1.3680043.10001_4.jpg')\nim = Image.open(fn)\nplt.imshow(im)","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:41.355791Z","iopub.execute_input":"2022-09-26T23:52:41.356169Z","iopub.status.idle":"2022-09-26T23:52:41.693Z","shell.execute_reply.started":"2022-09-26T23:52:41.356135Z","shell.execute_reply":"2022-09-26T23:52:41.691246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get the Z order of the images and sort from head to toe","metadata":{}},{"cell_type":"code","source":"def get_meta_zpos(fn):\n    ds = pydicom.dcmread(fn, stop_before_pixels=True)\n    if 'ImagePositionPatient' not in ds:\n        print('no position')\n    pos = ds.ImagePositionPatient[2]\n    return pos\n\n# get z position of first and last slice and compare, largest is more cranial\ndef add_order(df):\n    grouping = df.groupby('study_id')['slice']\n    order_seq = []\n    for study_id,g in grouping:\n        fn1 = get_fn(study_id, g.values[0])\n        fn2 = get_fn(study_id, g.values[-1])\n        pos1 = get_meta_zpos(fn1)\n        pos2 = get_meta_zpos(fn2)\n\n        #head first\n        if pos1 > pos2:\n            order = [o + 1 for o in range(len(g))]\n        else:\n            order = [o + 1 for o in reversed(range(len(g)))]\n        order_seq += order\n    return order_seq\n\nNEEDS_CSV = False\nif NEEDS_CSV == True:\n    #all train dcm filenames\n    tdf = pd.read_csv('../input/cspine-csvs/train_dcm_filenames.csv')\n    tdf = tdf.sort_values(['study_id','slice'])\n\n    #add order from head to thorax\n    tdf['order'] = add_order(tdf)\n    print(tdf.tail())\n    tdf.to_csv('train_dcm_filenames_ordered.csv',index=False)\nelse:\n    tdf = pd.read_csv(csv_root/'train_dcm_filenames_ordered.csv') ","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:41.694636Z","iopub.execute_input":"2022-09-26T23:52:41.695369Z","iopub.status.idle":"2022-09-26T23:52:42.259993Z","shell.execute_reply.started":"2022-09-26T23:52:41.695327Z","shell.execute_reply":"2022-09-26T23:52:42.258574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdf[(tdf.slice != tdf.order)].study_id.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:42.261667Z","iopub.execute_input":"2022-09-26T23:52:42.262033Z","iopub.status.idle":"2022-09-26T23:52:42.287787Z","shell.execute_reply.started":"2022-09-26T23:52:42.262Z","shell.execute_reply":"2022-09-26T23:52:42.286131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_id = '1.2.826.0.1.3680043.10001'","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:42.293649Z","iopub.execute_input":"2022-09-26T23:52:42.294124Z","iopub.status.idle":"2022-09-26T23:52:42.300653Z","shell.execute_reply.started":"2022-09-26T23:52:42.294083Z","shell.execute_reply":"2022-09-26T23:52:42.29911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transform resized and padded masks back to the image original shape","metadata":{}},{"cell_type":"code","source":"def get_padding(img):\n    r,c = img.shape\n    pad = int((abs(c-r)/2))\n    if img.shape[1] > img.shape[0]:\n        top, right, bottom, left = pad,0,abs(c-r)- pad,0\n    else:\n        top, right, bottom, left  = 0,pad,0,abs(c-r) - pad \n    return top, right, bottom, left \n\n#reverse image transforms from model prediction\ndef image_and_mask(fn):\n    #image\n    img = mpimg.imread(fn.with_suffix('.jpg'))\n    #get padding to be removed from mask\n    top, right, bottom, left = get_padding(img)\n    #mask - resize then remove padding\n    mask = np.load(fn.with_suffix('.npy'))\n    square_size = (max(img.shape),max(img.shape))\n    mask = resize(mask, square_size, anti_aliasing=True, preserve_range=True)\n    mask = mask.astype(int)\n    r,c = mask.shape\n    mask = mask[top:r - bottom, left:c - right]\n    return img, mask\n","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:42.302666Z","iopub.execute_input":"2022-09-26T23:52:42.303256Z","iopub.status.idle":"2022-09-26T23:52:42.316434Z","shell.execute_reply.started":"2022-09-26T23:52:42.303206Z","shell.execute_reply":"2022-09-26T23:52:42.314752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Example of a case that's not centered","metadata":{}},{"cell_type":"code","source":"fn = get_fn('1.2.826.0.1.3680043.11401',250)\nds = pydicom.dcmread(fn)\nim = ds.pixel_array\nplt.imshow(im)","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:42.318162Z","iopub.execute_input":"2022-09-26T23:52:42.319353Z","iopub.status.idle":"2022-09-26T23:52:42.618916Z","shell.execute_reply.started":"2022-09-26T23:52:42.31931Z","shell.execute_reply":"2022-09-26T23:52:42.617757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#8693, 11401\nshow_case('1.2.826.0.1.3680043.11401', 1.5)","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:42.620574Z","iopub.execute_input":"2022-09-26T23:52:42.621045Z","iopub.status.idle":"2022-09-26T23:52:43.955695Z","shell.execute_reply.started":"2022-09-26T23:52:42.620999Z","shell.execute_reply":"2022-09-26T23:52:43.954456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#typical case\nshow_case('1.2.826.0.1.3680043.10001', 1.5)","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:43.957124Z","iopub.execute_input":"2022-09-26T23:52:43.957494Z","iopub.status.idle":"2022-09-26T23:52:45.434464Z","shell.execute_reply.started":"2022-09-26T23:52:43.95746Z","shell.execute_reply":"2022-09-26T23:52:45.433317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get axial indices from sagittal slices\n\nGet the range of slices containing cspine (labels 1-7)\nMake sure the cspine is completely represented in the mask prediction, otherwise use the entire study","metadata":{}},{"cell_type":"code","source":"def get_mask_range(mask):\n    rows = [i for i, row in enumerate(mask) if row.max() > 0]\n    if len(rows) == 0:\n        return 0, mask.shape[0]\n    else:\n        return min(rows), max(rows)\n    \ndef get_index_from_slice(study_id, slice):\n    #get index by comparing slices to sort order\n    index = tdf[(tdf.study_id == study_id) & (tdf.slice == slice)].values[0][-1]\n    return index\n\ndef get_range_from_nii(study_id):\n    #read from range df created from segmentation files to get slice names\n    _,top,bottom = rdf[rdf.study_id == study_id].values[0]\n    #get index by comparing slices to sort order\n    top_index = get_index_from_slice(study_id, top)\n    bottom_index = get_index_from_slice(study_id, bottom)\n    return top_index,bottom_index\n    \ndef get_range_of_included(study_id, n_masks = 8):\n    if study_id in seg_ids:\n        top, bottom = get_range_from_nii(study_id)\n    else:\n        #get all masks\n        mask_fns = jpg_root.glob(study_id + '_*.npy')\n        if n_masks == 4:\n            tmp = []\n            sag_slices = [0,3,5,7]\n            for fn in mask_fns:\n                slice = int(fn.stem.split('_')[1])\n                if slice in sag_slices:\n                    tmp.append(fn)\n            if len(tmp) == 0:\n                print('error')\n            mask_fns = tmp\n        top = 0\n        bottom = 100000\n        #check if all cspine imaged\n        labels = { 1, 2, 3, 4, 5, 6, 7}\n        c = set()\n        for fn in mask_fns:\n            _,mask = image_and_mask(fn)\n            one,two = get_mask_range(mask)\n            x = set(mask.flatten())\n            c.update(x)\n            top = max(top,one)\n            bottom = min(bottom,two)\n        if c.intersection(labels) != labels:\n            print('Incomplete mask', c, study_id)\n            return 0, mask.shape[0]\n    return top, bottom","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:45.436214Z","iopub.execute_input":"2022-09-26T23:52:45.43665Z","iopub.status.idle":"2022-09-26T23:52:45.453178Z","shell.execute_reply.started":"2022-09-26T23:52:45.436612Z","shell.execute_reply":"2022-09-26T23:52:45.452308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_id = '1.2.826.0.1.3680043.10001'","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:45.454773Z","iopub.execute_input":"2022-09-26T23:52:45.455407Z","iopub.status.idle":"2022-09-26T23:52:45.470863Z","shell.execute_reply.started":"2022-09-26T23:52:45.455368Z","shell.execute_reply":"2022-09-26T23:52:45.469499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get list of images to keep based on results using 8 sagittal slices per study","metadata":{}},{"cell_type":"code","source":"# add keep column\nfn = Path('../input/cspine-csvs/keep_8_seg_pred.csv')\nif fn.exists():\n    tdf = pd.read_csv(fn)\nelse:\n    grouped = tdf.groupby('study_id')\n\n    keep_list = []\n    for study_id, g in grouped:\n        top, bottom = get_range_of_included(study_id)\n        items = g.order.tolist()\n        keep = [((int(item) >= top) and (int(item) <= bottom)) for item in items]\n\n        keep_list.extend(keep)\n\n    tdf['keep_8'] = keep_list\n    tdf.to_csv('keep_8_seg_pred.csv',index=False)\n    tdf.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:45.473017Z","iopub.execute_input":"2022-09-26T23:52:45.473538Z","iopub.status.idle":"2022-09-26T23:52:46.067521Z","shell.execute_reply.started":"2022-09-26T23:52:45.473491Z","shell.execute_reply":"2022-09-26T23:52:46.066337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdf.keep.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:46.068818Z","iopub.execute_input":"2022-09-26T23:52:46.069204Z","iopub.status.idle":"2022-09-26T23:52:46.085091Z","shell.execute_reply.started":"2022-09-26T23:52:46.069168Z","shell.execute_reply":"2022-09-26T23:52:46.083856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get list of images to keep based on results using only 4 sagittal slices per study","metadata":{}},{"cell_type":"code","source":"# evaluate to see if 4 images will get close to the same results\nfn = Path('../input/cspine-csvs/keep_4_seg_pred.csv')\nif fn.exists():\n    tdf = pd.read_csv(fn)\nelse:\n    grouped = tdf.groupby('study_id')\n\n    keep_list = []\n    for study_id, g in grouped:\n        top, bottom = get_range_of_included(study_id, n_masks=4)\n        items = g.order.tolist()\n        keep = [((int(item) >= top) and (int(item) <= bottom)) for item in items]\n\n        keep_list.extend(keep)\n\n    tdf['keep_4'] = keep_list\n    tdf.to_csv('keep_4_seg_pred.csv',index=False)\n    tdf.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:46.08679Z","iopub.execute_input":"2022-09-26T23:52:46.087559Z","iopub.status.idle":"2022-09-26T23:52:46.774198Z","shell.execute_reply.started":"2022-09-26T23:52:46.08752Z","shell.execute_reply":"2022-09-26T23:52:46.772888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdf.keep_4.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:46.775499Z","iopub.execute_input":"2022-09-26T23:52:46.775829Z","iopub.status.idle":"2022-09-26T23:52:46.790251Z","shell.execute_reply.started":"2022-09-26T23:52:46.775799Z","shell.execute_reply":"2022-09-26T23:52:46.789303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Keep list from prior notebook which made predictions on individual images","metadata":{}},{"cell_type":"code","source":"#compare to rdf\nfn = Path('../input/cspine-csvs/keep_variations.csv')\nif fn.exists():\n    tdf = pd.read_csv(fn)\nelse:\n    grouped = tdf.groupby('study_id')\n\n    keep_list = []\n    for study_id, g in grouped:\n        #ranges df\n        row = rdf[rdf.study_id == study_id].values[0]\n        #print(row, row[2] + 1 - row[1])\n        top = row[1]\n        bottom = row[2]\n        items = g.slice.tolist()\n        #print(len(items))\n        keep = [((int(item) >= top) and (int(item) <= bottom)) for item in items]\n        #print(len([o for o in keep if o == True]))\n        assert row[2] + 1 - row[1] == len([o for o in keep if o == True]), f'Discrepancy {study_id}'\n        keep_list.extend(keep)\n\n    tdf['keep_slice_pred'] = keep_list\n    tdf.columns = ['study_id', 'slice', 'order', 'keep_8', 'keep_4', 'keep_slice_pred']\n    tdf.to_csv('keep_variations.csv',index=False)\n ","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:46.791762Z","iopub.execute_input":"2022-09-26T23:52:46.792851Z","iopub.status.idle":"2022-09-26T23:52:47.549085Z","shell.execute_reply.started":"2022-09-26T23:52:46.79281Z","shell.execute_reply":"2022-09-26T23:52:47.548148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdf.keep_slice_pred.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:47.550225Z","iopub.execute_input":"2022-09-26T23:52:47.550802Z","iopub.status.idle":"2022-09-26T23:52:47.566268Z","shell.execute_reply.started":"2022-09-26T23:52:47.550765Z","shell.execute_reply":"2022-09-26T23:52:47.564837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Comparison\n\nUsing only 4 images gives similar results. 3751 (523244 - 519493) extra images are kept compared to using 8 sagittal slices. Either way, you would want to give a top and bottom margin and include extra images.\n\nThe slices bases on the notebook 'CSpine Reduce Data by 1/3' has the tightest predictions with more excluded slices. The training took longer and gave no information on CSpine levels.\n\n## Final assessment:\n\nPredictions with 4 sagittal images spaced at 24 pixels apart gave the most info in the shortest amount of time","metadata":{}},{"cell_type":"code","source":"# show mask images for include from all 3 methods\n\n#reverse image transforms from model prediction\ndef transform_mask(fn, top, bottom):\n    #mask - resize then remove rows\n    mask = np.load(fn.with_suffix('.npy'))\n    square_size = (max(img.shape),max(img.shape))\n    mask = resize(mask, square_size, anti_aliasing=True, preserve_range=True)\n    mask = mask.astype(int)\n    r,c = mask.shape\n    mask = mask[top:r - bottom, left:c - right]\n    return mask","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:47.567768Z","iopub.execute_input":"2022-09-26T23:52:47.568799Z","iopub.status.idle":"2022-09-26T23:52:47.577977Z","shell.execute_reply.started":"2022-09-26T23:52:47.568761Z","shell.execute_reply":"2022-09-26T23:52:47.576885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdf.columns","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:47.579605Z","iopub.execute_input":"2022-09-26T23:52:47.579951Z","iopub.status.idle":"2022-09-26T23:52:47.58952Z","shell.execute_reply.started":"2022-09-26T23:52:47.579921Z","shell.execute_reply":"2022-09-26T23:52:47.588403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_range_for_column(column, study_id):\n    study = tdf[(tdf.study_id == study_id)]\n    keep_indices = [row['order'] for _,row in study.iterrows() if row[column] == True]\n    if len(keep_indices) == 0:\n        print('keep none', study_id)\n        return 1,len(study)\n    top_index = min(keep_indices)\n    bottom_index = max(keep_indices)\n    return top_index, bottom_index\n    \n    \ndef show_keep(study_ids, slice = 3, aspect = 2):\n    fig, axs = plt.subplots(6,3, figsize=(12, 16))\n    axs = axs.flatten()\n    for i in range(0, len(axs), 3):\n        study_id = study_ids[int(i/3)]\n        fn = get_fn(study_id, slice, '.jpg')\n\n#         assert img.shape[0] ==  count_df[count_df.study_id== study_id].values[0][1], 'Shape does not match'\n        \n        columns = tdf.columns[-3:] #['keep', 'keep_4', 'keep_slice_pred']\n        for j, column in enumerate(columns):\n            top,bottom = get_range_for_column(column, study_id)\n            img = mpimg.imread(fn)\n            img = img[top: bottom + 1]\n            axs[i + j].imshow(img, cmap='bone')\n            axs[i + j].axis(\"off\")\n            axs[i + j].set_title(f'{column} ({top}-{bottom})')\n            axs[i + j].set_aspect(aspect)\n        \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:47.591337Z","iopub.execute_input":"2022-09-26T23:52:47.591679Z","iopub.status.idle":"2022-09-26T23:52:47.603724Z","shell.execute_reply.started":"2022-09-26T23:52:47.59165Z","shell.execute_reply":"2022-09-26T23:52:47.602183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#show_ids = study_ids[80:86]\nstudy_ids = np.array(study_ids)\nselector = np.array([100, 20,30, 1889, 500,550],int)\n#ind = [100, 20,30, 300, 500,550]\nshow_ids = study_ids[selector]\nshow_keep(show_ids)","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:47.605254Z","iopub.execute_input":"2022-09-26T23:52:47.605615Z","iopub.status.idle":"2022-09-26T23:52:50.392068Z","shell.execute_reply.started":"2022-09-26T23:52:47.605585Z","shell.execute_reply":"2022-09-26T23:52:50.39085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = rdf.copy()\ndf.columns = ['study_id','top_keep_slice_pred', 'bottom_keep_slice_pred']\n#df = df.set_index('study_id')\ndf.head()        ","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:50.393723Z","iopub.execute_input":"2022-09-26T23:52:50.394215Z","iopub.status.idle":"2022-09-26T23:52:50.408421Z","shell.execute_reply.started":"2022-09-26T23:52:50.394171Z","shell.execute_reply":"2022-09-26T23:52:50.407386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fn = Path('../input/cspine-csvs/ranges.csv')\nif fn.exists():\n    df = pd.read_csv(fn)\nelse:\n    print('Retrieving range_8')\n    df['range_8'] = [get_range_for_column('keep', study_id) for study_id in df.study_id.unique()]\n    print('Retrieving range_4')\n    df['range_4'] = [get_range_for_column('keep_4', study_id) for study_id in df.study_id.unique()]\n    print('Separate columns')\n    df['top_8'] = [item[0] for _,item in df.range_8.iteritems()]\n    df['bottom_8'] = [item[1] for _,item in df.range_8.iteritems()]\n    df['top_4'] = [item[0] for _,item in df.range_4.iteritems()]\n    df['bottom_4'] = [item[1] for _,item in df.range_4.iteritems()]\n    df.drop(['range_8','range_4'], axis=1,inplace=True)\n    df.to_csv('ranges.csv',index=False)\n\n#add count\ndf = df.merge(count_df)\n#add direction from df that has slice and order\nhead_first = []\ngrouped = tdf.groupby('study_id')\nfor study_id, g in grouped:\n    head_first.append((abs(g['slice'].values[0] - g['order'].values[0])) < 10)\n\ndirection_df = pd.DataFrame({'study_id':tdf.study_id.unique(), \n                             'head_first':head_first})\ndf = df.merge(direction_df)\n# add slice thickness\nmeta = pd.read_csv('../input/cspine-csvs/train_metadata.csv')\nmeta = meta[[ 'StudyInstanceUID','SliceThickness']]\nmeta.columns = ['study_id', 'thickness']\ndf = df.merge(meta)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:13:37.004598Z","iopub.execute_input":"2022-09-27T00:13:37.005156Z","iopub.status.idle":"2022-09-27T00:13:37.503107Z","shell.execute_reply.started":"2022-09-27T00:13:37.005109Z","shell.execute_reply":"2022-09-27T00:13:37.501827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#keep_8\n# keep none 1.2.826.0.1.3680043.16729\n# keep none 1.2.826.0.1.3680043.30307\n# keep none 1.2.826.0.1.3680043.3593\n#keep_4\n# keep none 1.2.826.0.1.3680043.30307\n# keep none 1.2.826.0.1.3680043.3593","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:13:38.373033Z","iopub.execute_input":"2022-09-27T00:13:38.373773Z","iopub.status.idle":"2022-09-27T00:13:38.378467Z","shell.execute_reply.started":"2022-09-27T00:13:38.373735Z","shell.execute_reply":"2022-09-27T00:13:38.376958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Off center\nshow_case('1.2.826.0.1.3680043.3593', 1)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:13:38.654371Z","iopub.execute_input":"2022-09-27T00:13:38.654751Z","iopub.status.idle":"2022-09-27T00:13:40.395267Z","shell.execute_reply.started":"2022-09-27T00:13:38.654721Z","shell.execute_reply":"2022-09-27T00:13:40.393706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Patient in decubitus position\nshow_case('1.2.826.0.1.3680043.16729', 1)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:13:40.397695Z","iopub.execute_input":"2022-09-27T00:13:40.399554Z","iopub.status.idle":"2022-09-27T00:13:41.784016Z","shell.execute_reply.started":"2022-09-27T00:13:40.399485Z","shell.execute_reply":"2022-09-27T00:13:41.782714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fn = get_fn('1.2.826.0.1.3680043.16729',250)\nds = pydicom.dcmread(fn)\nim = ds.pixel_array\nplt.imshow(im)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:13:41.785416Z","iopub.execute_input":"2022-09-27T00:13:41.785784Z","iopub.status.idle":"2022-09-27T00:13:42.066306Z","shell.execute_reply.started":"2022-09-27T00:13:41.78575Z","shell.execute_reply":"2022-09-27T00:13:42.065137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df.study_id == '1.2.826.0.1.3680043.16729']","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:13:42.069337Z","iopub.execute_input":"2022-09-27T00:13:42.070165Z","iopub.status.idle":"2022-09-27T00:13:42.088851Z","shell.execute_reply.started":"2022-09-27T00:13:42.070115Z","shell.execute_reply":"2022-09-27T00:13:42.088055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df.hist('top_8', bins=40), df.hist('top_4', bins=40),df.hist('top_keep_slice_pred', bins=20)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:13:42.090556Z","iopub.execute_input":"2022-09-27T00:13:42.091347Z","iopub.status.idle":"2022-09-27T00:13:42.096472Z","shell.execute_reply.started":"2022-09-27T00:13:42.091301Z","shell.execute_reply":"2022-09-27T00:13:42.095185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df.hist('bottom_8', bins=40), df.hist('bottom_4', bins=40),df.hist('bottom_keep_slice_pred', bins=20)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:13:42.098223Z","iopub.execute_input":"2022-09-27T00:13:42.098596Z","iopub.status.idle":"2022-09-27T00:13:42.106843Z","shell.execute_reply.started":"2022-09-27T00:13:42.098523Z","shell.execute_reply":"2022-09-27T00:13:42.10563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#total slices to keep\ndf['number_4'] = df.bottom_4 - df.top_4\ndf['number_slice'] = df.bottom_keep_slice_pred - df.top_keep_slice_pred\n\n#identify problem cases by ratio of keep to total\ndf['percent'] = df['number_4']/df.n\ndf.percent.hist(bins=40)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:14:40.174515Z","iopub.execute_input":"2022-09-27T00:14:40.175268Z","iopub.status.idle":"2022-09-27T00:14:40.472313Z","shell.execute_reply.started":"2022-09-27T00:14:40.175216Z","shell.execute_reply":"2022-09-27T00:14:40.471014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"problems = df[(df.number_4/df.number_slice < .50)]\nprint(len(problems))\nproblems.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:13:42.484198Z","iopub.execute_input":"2022-09-27T00:13:42.48535Z","iopub.status.idle":"2022-09-27T00:13:42.510429Z","shell.execute_reply.started":"2022-09-27T00:13:42.485299Z","shell.execute_reply":"2022-09-27T00:13:42.509116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#show some problem cases\nids = np.array(df.study_id.tolist())\nproblems = problems.index\nshow_ids = ids[problems[:8]]\nshow_keep(show_ids)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:13:43.272635Z","iopub.execute_input":"2022-09-27T00:13:43.273294Z","iopub.status.idle":"2022-09-27T00:13:45.579551Z","shell.execute_reply.started":"2022-09-27T00:13:43.273247Z","shell.execute_reply":"2022-09-27T00:13:45.578688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:13:45.581117Z","iopub.execute_input":"2022-09-27T00:13:45.581681Z","iopub.status.idle":"2022-09-27T00:13:45.599114Z","shell.execute_reply.started":"2022-09-27T00:13:45.581644Z","shell.execute_reply":"2022-09-27T00:13:45.598017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Percent of predictions over total images can be used to find cases where keep_4 falls short.","metadata":{}},{"cell_type":"code","source":"p = df[(df.percent < .35) & ((df.number_4/df.number_slice < .50))]\nprint(p.shape)\np.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:53:16.715916Z","iopub.execute_input":"2022-09-26T23:53:16.717262Z","iopub.status.idle":"2022-09-26T23:53:16.73788Z","shell.execute_reply.started":"2022-09-26T23:53:16.717211Z","shell.execute_reply":"2022-09-26T23:53:16.736321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions from slices are better at predicting C1\n\nAdd 10-15mm to the top of the keep_4 prediction to capture C1. If predictions are less than 35% of total slices, ignore and use all images.","metadata":{}},{"cell_type":"code","source":"df.plot(kind='hist', bins=40)","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:53:16.768445Z","iopub.execute_input":"2022-09-26T23:53:16.76958Z","iopub.status.idle":"2022-09-26T23:53:18.054396Z","shell.execute_reply.started":"2022-09-26T23:53:16.76952Z","shell.execute_reply":"2022-09-26T23:53:18.053071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get slice thickness\n# meta = pd.read_csv('../input/cspine-csvs/train_metadata.csv')\n# meta = meta[[ 'StudyInstanceUID','SliceThickness']]\n# meta.columns = ['study_id', 'thickness']\n# df = df.merge(meta)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:12:11.826631Z","iopub.execute_input":"2022-09-27T00:12:11.827285Z","iopub.status.idle":"2022-09-27T00:12:11.848293Z","shell.execute_reply.started":"2022-09-27T00:12:11.827232Z","shell.execute_reply":"2022-09-27T00:12:11.847098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df[['study_id', 'top_4', 'bottom_4', 'n', 'number_slice',\n       'number_4', 'percent', 'thickness', 'head_first']]\ndf.columns = ['study_id', 'top_4', 'bottom_4', 'total', 'diff_slice',\n       'diff_4', 'percent_of_total', 'thickness', 'head_first']\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:15:19.939817Z","iopub.execute_input":"2022-09-27T00:15:19.940241Z","iopub.status.idle":"2022-09-27T00:15:19.964473Z","shell.execute_reply.started":"2022-09-27T00:15:19.940205Z","shell.execute_reply":"2022-09-27T00:15:19.963581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# slices_to_add = [10/item for _,item in df.thickness.iteritems()]\n# df['slices_to_add'] = slices_to_add\n# df.slices_to_add = df.slices_to_add.astype(int)\n# df.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:17:08.096581Z","iopub.execute_input":"2022-09-27T00:17:08.097171Z","iopub.status.idle":"2022-09-27T00:17:08.134394Z","shell.execute_reply.started":"2022-09-27T00:17:08.097123Z","shell.execute_reply":"2022-09-27T00:17:08.131555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Final slices\n\nChange problem cases to entire range\nThere aren't that many cases (31) so it should not have a large impact.","metadata":{}},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:18:14.716963Z","iopub.execute_input":"2022-09-27T00:18:14.718438Z","iopub.status.idle":"2022-09-27T00:18:14.737549Z","shell.execute_reply.started":"2022-09-27T00:18:14.718376Z","shell.execute_reply":"2022-09-27T00:18:14.736047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"problems = p.index.tolist()\nfinal_df = df.copy()\nfor index in problems:\n    final_df.loc[index,'top_4'] = 1\n    final_df.loc[index,'bottom_4'] = final_df.loc[index,'total']\n    \nfinal_df.loc[72:170]","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:18:23.251063Z","iopub.execute_input":"2022-09-27T00:18:23.251485Z","iopub.status.idle":"2022-09-27T00:18:23.299891Z","shell.execute_reply.started":"2022-09-27T00:18:23.251449Z","shell.execute_reply":"2022-09-27T00:18:23.298416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Add padding to head","metadata":{}},{"cell_type":"code","source":"def new_top_bottom(row):\n    amt = int(10/row.thickness)\n    if head_first == True:\n        return max(0,row.top_4 - amt), row.bottom_4\n    else:\n        return row.top_4, min(row.bottom_4 + amt, row.total)\n    \nfinal_df['new_range'] = [new_top_bottom(row) for _,row in final_df.iterrows()]\nfinal_df['top'] = [item[0] for _,item in final_df.new_range.iteritems()]\nfinal_df['bottom'] = [item[1] for _,item in final_df.new_range.iteritems()]\nfinal_df = final_df[['study_id', 'top', 'bottom', 'total', 'thickness', 'head_first']]\nfinal_df['slice_diff'] = final_df.bottom - final_df.top\nfinal_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:27:31.729185Z","iopub.execute_input":"2022-09-27T00:27:31.729569Z","iopub.status.idle":"2022-09-27T00:27:31.793899Z","shell.execute_reply.started":"2022-09-27T00:27:31.729538Z","shell.execute_reply":"2022-09-27T00:27:31.792196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# final_df = final_df[['study_id', 'top', 'bottom', 'total', 'thickness', 'head_first']]\n# final_df['slice_diff'] = final_df.bottom - final_df.top\n# final_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:29:13.545583Z","iopub.execute_input":"2022-09-27T00:29:13.546316Z","iopub.status.idle":"2022-09-27T00:29:13.563439Z","shell.execute_reply.started":"2022-09-27T00:29:13.546277Z","shell.execute_reply":"2022-09-27T00:29:13.562237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df['slice_diff'].sum(), final_df.total.sum(), final_df['slice_diff'].sum()/ final_df.total.sum()","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:30:22.755905Z","iopub.execute_input":"2022-09-27T00:30:22.756321Z","iopub.status.idle":"2022-09-27T00:30:22.766077Z","shell.execute_reply.started":"2022-09-27T00:30:22.756288Z","shell.execute_reply":"2022-09-27T00:30:22.76466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df.diff_4.sum(),df.diff_4.sum()/final_df.total.sum(), df.diff_slice.sum(), df.diff_slice.sum()/ final_df.total.sum()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:53.88991Z","iopub.status.idle":"2022-09-26T23:52:53.890379Z","shell.execute_reply.started":"2022-09-26T23:52:53.890169Z","shell.execute_reply":"2022-09-26T23:52:53.890189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df.to_csv('range_from_4_with_10mm_head_padding.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-09-27T00:30:58.237377Z","iopub.execute_input":"2022-09-27T00:30:58.237799Z","iopub.status.idle":"2022-09-27T00:30:58.255111Z","shell.execute_reply.started":"2022-09-27T00:30:58.237762Z","shell.execute_reply":"2022-09-27T00:30:58.254153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T23:52:53.893672Z","iopub.status.idle":"2022-09-26T23:52:53.894119Z","shell.execute_reply.started":"2022-09-26T23:52:53.893885Z","shell.execute_reply":"2022-09-26T23:52:53.893904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}