{"cells":[{"metadata":{},"cell_type":"markdown","source":"# fastai starter\n\nMany thanks to [Basic EDA + Data Visualization 🧠 ](https://www.kaggle.com/marcovasquez/basic-eda-data-visualization) for the code to load the data."},{"metadata":{},"cell_type":"markdown","source":"## Imports"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import glob, pylab, pandas as pd\nimport pydicom, numpy as np\nfrom os import listdir\nfrom os.path import isfile, join\nimport matplotlib.pylab as plt\n\nimport seaborn as sns\n\nfrom tqdm import tqdm_notebook as tqdm\nfrom fastai.vision import *","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Load and preprocess data\n\nWe will transform the data into a nice space separated label format."},{"metadata":{"trusted":true},"cell_type":"code","source":"DATA = Path(\"../input/rsna-intracranial-hemorrhage-detection\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv('../input/rsna-intracranial-hemorrhage-detection/stage_1_train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"newtable = df.copy()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new = newtable[\"ID\"].str.split(\"_\", n = 1, expand = True)\nnewX = new[1].str.split(\"_\", n = 1, expand = True)\nnewX[1]\nnewtable['Image_ID'] = newX[0]\nnewtable['Sub_type'] = newX[1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_ids = newtable.Image_ID.unique()\nlabels = [\"\" for _ in range(len(image_ids))]\nnew_df = pd.DataFrame(np.array([image_ids, labels]).transpose(), columns=[\"id\", \"labels\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lbls = {i : \"\" for i in image_ids}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"newtable = newtable[newtable.Label == 1]\nnewtable = newtable[newtable.Sub_type != \"any\"]\n\ni = 0\nfor name, group in newtable.groupby(\"Image_ID\"):\n    lbls[name] = \" \".join(group.Sub_type)\n    if i % 10000 == 0: print(i)\n    i += 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_df = pd.DataFrame(np.array([list(lbls.keys()), list(lbls.values())]).transpose(), columns=[\"id\", \"labels\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del lbls\ndel newtable\ndel newX\ndel new\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# fastai Dataset\n\nThanks to this kernel for the code to apply the windowing: [EDA: View dicom images with correct windowing](https://www.kaggle.com/omission/eda-view-dicom-images-with-correct-windowing)"},{"metadata":{"trusted":true},"cell_type":"code","source":"#https://www.kaggle.com/omission/eda-view-dicom-images-with-correct-windowing\n\ndef window_image(img, window_center,window_width, intercept, slope):\n\n    img = (img*slope +intercept)\n    img_min = window_center - window_width//2\n    img_max = window_center + window_width//2\n    img[img<img_min] = img_min\n    img[img>img_max] = img_max\n    return img\n\ndef get_first_of_dicom_field_as_int(x):\n    #get x[0] as in int is x is a 'pydicom.multival.MultiValue', otherwise get int(x)\n    if type(x) == pydicom.multival.MultiValue:\n        return int(x[0])\n    else:\n        return int(x)\n\ndef get_windowing(data):\n    dicom_fields = [data[('0028','1050')].value, #window center\n                    data[('0028','1051')].value, #window width\n                    data[('0028','1052')].value, #intercept\n                    data[('0028','1053')].value] #slope\n    return [get_first_of_dicom_field_as_int(x) for x in dicom_fields]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_df.id = \"ID_\" + new_df.id + \".dcm\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def new_open_image(path, div=True, convert_mode=None, after_open=None):\n    dcm = pydicom.dcmread(str(path))\n    window_center, window_width, intercept, slope = get_windowing(dcm)\n    im = window_image(dcm.pixel_array, window_center, window_width, intercept, slope)\n    im = np.stack((im,)*3, axis=-1)\n    im -= im.min()\n    im_max = im.max()\n    if im_max != 0: im = im / im.max()\n    x = Image(pil2tensor(im, dtype=np.float32))\n    #if div: x.div_(2048)  # ??\n    return x\n\n\nvision.data.open_image = new_open_image","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We will train on a subset of the data so that it doesn't take too long: 15000 examples containing one or more hemorrhage types, 15000 examples containing no hemorrhages"},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = pd.concat([new_df[new_df.labels == \"\"][:15000], new_df[new_df.labels != \"\"][:15000]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bs = 128\n\nim_list = ImageList.from_df(df_train, path=DATA/\"stage_1_train_images\")\ntest_fnames = pd.DataFrame(\"ID_\" + pd.read_csv(DATA/\"stage_1_sample_submission.csv\")[\"ID\"].str.split(\"_\", n=2, expand = True)[1].unique() + \".dcm\")\ntest_im_list = ImageList.from_df(test_fnames, path=DATA/\"stage_1_test_images\")\n\ntfms = get_transforms(do_flip=False)\n\ndata = (im_list.split_by_rand_pct(0.2)\n               .label_from_df(label_delim=\" \")\n               .transform(tfms, size=512)\n               .add_test(test_im_list)\n               .databunch(bs=bs, num_workers=0)\n               .normalize())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.show_batch(3)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Train the model\n\nA resnet18 with imagenet weights."},{"metadata":{"trusted":true},"cell_type":"code","source":"learn = cnn_learner(data, models.resnet18)\n\nmodels_path = Path(\"/kaggle/working/models\")\nif not models_path.exists(): models_path.mkdir()\n    \nlearn.model_dir = models_path\nlearn.metrics = [accuracy_thresh]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.lr_find()\nlearn.recorder.plot()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.fit_one_cycle(4, 5e-2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.unfreeze()\nlearn.lr_find()\nlearn.recorder.plot()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.fit_one_cycle(12, slice(1e-3))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Submission\n\nThe predicted probability for **any** hemorrhage being present is calculated as $1 - ((1 - \\text{P}(\\text{type 1})) \\times (1 - \\text{P}(\\text{type 2})) \\dots) $"},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = pd.read_csv(DATA/\"stage_1_sample_submission.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = learn.get_preds(ds_type=DatasetType.Test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = np.array(preds[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"any_probs = 1 - np.prod(1 - preds, axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"any_probs.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.Label = np.hstack([preds, np.expand_dims(any_probs, -1)]).reshape(-1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Good luck everyone!"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}