{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!mkdir conda-pkgs\n # Set the location where conda package will be downloaded\n!conda config --add pkgs_dirs ./conda-pkgs\n # Download pyvips and dependencies\n!conda install --download-only -y \"pyvips>=2.2.0\"\n!rm -rf ./conda-pkgs/cache\n!conda install ./conda-pkgs/*.tar.bz2 --offline","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:59:34.963997Z","iopub.execute_input":"2022-08-11T06:59:34.964452Z","iopub.status.idle":"2022-08-11T07:02:20.100513Z","shell.execute_reply.started":"2022-08-11T06:59:34.964419Z","shell.execute_reply":"2022-08-11T07:02:20.098527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport cv2\nimport pyvips\nfrom numba import njit\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_squared_error\n\nimport skimage\nfrom skimage.filters import sobel\nfrom skimage import segmentation\nfrom skimage.color import label2rgb\nfrom skimage.color import rgb2hed, hed2rgb\nfrom skimage.exposure import rescale_intensity\nimport tifffile as tifi\n\nfrom PIL import Image, ImageOps, ImageFilter\nImage.MAX_IMAGE_PIXELS = None\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T08:25:12.429628Z","iopub.execute_input":"2022-08-11T08:25:12.430138Z","iopub.status.idle":"2022-08-11T08:25:12.448247Z","shell.execute_reply.started":"2022-08-11T08:25:12.430086Z","shell.execute_reply":"2022-08-11T08:25:12.44716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Are we still using Lib PIL for images. Please optimize and remove PythonImageLibrary","metadata":{}},{"cell_type":"code","source":"path = '../input/mayo-clinic-strip-ai/'","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:25:15.894911Z","iopub.execute_input":"2022-08-11T08:25:15.895321Z","iopub.status.idle":"2022-08-11T08:25:15.902948Z","shell.execute_reply.started":"2022-08-11T08:25:15.895289Z","shell.execute_reply":"2022-08-11T08:25:15.901683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('{}/train.csv'.format(path))\ndf['path'] = path + 'train/' + df['image_id'] + '.tif'\ndf['target'] = df['label'].map({'CE': 0, 'LAA': 1})","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:27:09.392672Z","iopub.execute_input":"2022-08-10T10:27:09.393035Z","iopub.status.idle":"2022-08-10T10:27:09.441118Z","shell.execute_reply.started":"2022-08-10T10:27:09.393002Z","shell.execute_reply":"2022-08-10T10:27:09.43989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['shape'] = df['path'].map(lambda x: Image.open(x).size)\ndf[['width', 'height']] = pd.DataFrame(df['shape'].tolist(), columns=['width', 'height'])\ndf['area'] = df['width'] * df['height']","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:27:09.442956Z","iopub.execute_input":"2022-08-10T10:27:09.443715Z","iopub.status.idle":"2022-08-10T10:27:26.072081Z","shell.execute_reply.started":"2022-08-10T10:27:09.443677Z","shell.execute_reply":"2022-08-10T10:27:26.071107Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:27:26.07323Z","iopub.execute_input":"2022-08-10T10:27:26.073548Z","iopub.status.idle":"2022-08-10T10:27:26.101644Z","shell.execute_reply.started":"2022-08-10T10:27:26.073519Z","shell.execute_reply":"2022-08-10T10:27:26.100434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['area'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:27:26.103474Z","iopub.execute_input":"2022-08-10T10:27:26.10421Z","iopub.status.idle":"2022-08-10T10:27:26.119107Z","shell.execute_reply.started":"2022-08-10T10:27:26.104157Z","shell.execute_reply":"2022-08-10T10:27:26.117802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['area'].plot(kind='hist', grid=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:27:26.120492Z","iopub.execute_input":"2022-08-10T10:27:26.121389Z","iopub.status.idle":"2022-08-10T10:27:26.38552Z","shell.execute_reply.started":"2022-08-10T10:27:26.121323Z","shell.execute_reply":"2022-08-10T10:27:26.384244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(x='width', y='height', hue='label', data=df)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:27:26.386855Z","iopub.execute_input":"2022-08-10T10:27:26.387173Z","iopub.status.idle":"2022-08-10T10:27:26.68295Z","shell.execute_reply.started":"2022-08-10T10:27:26.387144Z","shell.execute_reply":"2022-08-10T10:27:26.682157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#img_path = df[df['width'].gt(df['height'])].iloc[9]['path']\nimg_path = df[df['width'].gt(df['height'])].sample(1).iloc[0]['path']\nimg = pyvips.Image.thumbnail(img_path, 20000).numpy()\nplt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:27:26.686337Z","iopub.execute_input":"2022-08-10T10:27:26.686912Z","iopub.status.idle":"2022-08-10T10:27:54.944701Z","shell.execute_reply.started":"2022-08-10T10:27:26.686877Z","shell.execute_reply":"2022-08-10T10:27:54.94342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tile(img, sz=1024):\n    shape = img.shape\n    img_mean = img.mean()\n    pad0,pad1 = (sz - shape[0]%sz)%sz, (sz - shape[1]%sz)%sz\n    img = np.pad(img,[[pad0//2,pad0-pad0//2],[pad1//2,pad1-pad1//2],[0,0]],constant_values=255)\n    img = img.reshape(img.shape[0]//sz,sz,img.shape[1]//sz,sz,3)\n    img = img.transpose(0,2,1,3,4).reshape(-1,sz,sz,3)\n    idxs = [i.mean() < img_mean for i in img]\n    img = img[idxs]\n    return img","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:06:27.32396Z","iopub.execute_input":"2022-08-11T07:06:27.325184Z","iopub.status.idle":"2022-08-11T07:06:27.334302Z","shell.execute_reply.started":"2022-08-11T07:06:27.325138Z","shell.execute_reply":"2022-08-11T07:06:27.333094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nimg = tile(img, sz=2048)\nprint(nimg.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:27:54.957213Z","iopub.execute_input":"2022-08-10T10:27:54.957619Z","iopub.status.idle":"2022-08-10T10:27:56.49645Z","shell.execute_reply.started":"2022-08-10T10:27:54.957584Z","shell.execute_reply":"2022-08-10T10:27:56.49518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row = min(int(nimg.shape[0]**0.5)+1, 12)\ncol = row\n\nfig, axes = plt.subplots(row, col, figsize=(18, 12))\nfor idx, ax in enumerate(axes.flatten()):\n    try:\n        ax.imshow(nimg[idx])\n        ax.set_title('Good')\n        ax.set(xticklabels=[], yticklabels=[], xticks=[], yticks=[])\n    except:\n        ax.imshow(np.ones((1024, 1024, 3)))\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:27:56.497931Z","iopub.execute_input":"2022-08-10T10:27:56.498278Z","iopub.status.idle":"2022-08-10T10:28:04.676242Z","shell.execute_reply.started":"2022-08-10T10:27:56.498248Z","shell.execute_reply":"2022-08-10T10:28:04.675037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# look into numba\ndef prune_image_rows_cols(im, mask, thr=0.990):\n    # delete empty columns\n    for l in range(im.shape[1]):\n        if (np.sum(mask[:, l]) / float(mask.shape[0])) > thr:\n            im = np.delete(im, l, 1)\n    # delete empty rows\n    for l in reversed(range(im.shape[0])):\n        if (np.sum(mask[l, :]) / float(mask.shape[1])) > thr:\n            im = np.delete(im, l, 0)\n    return im\n\ndef mask_median(im, val=255):\n    masks = [None] * 3\n    for c in range(3):\n        masks[c] = im[..., c] >= np.median(im[:, :, c]) - 5\n    mask = np.logical_and(*masks)\n    im[mask, :] = val\n    return im, mask\n\n#fig, axes = plt.subplots(nrows=1, ncols=2, figsize=(14, 10))\n#img_path = df.iloc[df[\"area\"].idxmax(),:]['path']\n#img_path = df.sample(1)['path'].iloc[0]\n#img_path = df[df['image_id'].eq(filename)]['path'].iloc[0]\n#img = pyvips.Image.thumbnail(img_path, 20000).numpy()\n#if img.shape[0] > img.shape[1]:\n    #img = np.rollaxis(img, 0, 2)\n#axes[0].imshow(img)\n\n#img, mask = mask_median(np.array(img))\n#img = prune_image_rows_cols(img, mask)\n#axes[1].imshow(img)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:06:37.379739Z","iopub.execute_input":"2022-08-11T07:06:37.38024Z","iopub.status.idle":"2022-08-11T07:06:37.391322Z","shell.execute_reply.started":"2022-08-11T07:06:37.380199Z","shell.execute_reply":"2022-08-11T07:06:37.390172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_images(df):\n    # detect edges and remove empty spaces\n    rows = 2\n    cols = 2\n    fig, ax = plt.subplots(rows, cols, figsize=(14, 8))\n    for idx, i in enumerate(ax.flatten()):\n        row = df.sample(1).iloc[0]\n        img_path, label = row['path'], row['label']\n        img = pyvips.Image.thumbnail(img_path, 2000).numpy()\n        if img.shape[0] > img.shape[1]:\n            img = np.rollaxis(img, 0, 2)\n        img, mask = mask_median(np.array(img))\n        img = prune_image_rows_cols(img, mask)\n        \n        i.imshow(img)\n        i.set_title('{}-{}'.format(label, img_path.split('/')[-1]))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:06:42.128165Z","iopub.execute_input":"2022-08-11T07:06:42.129221Z","iopub.status.idle":"2022-08-11T07:06:42.138755Z","shell.execute_reply.started":"2022-08-11T07:06:42.129175Z","shell.execute_reply":"2022-08-11T07:06:42.137396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot_images(df[df['label'] == 'CE'])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:28:04.707519Z","iopub.execute_input":"2022-08-10T10:28:04.708771Z","iopub.status.idle":"2022-08-10T10:28:04.715428Z","shell.execute_reply.started":"2022-08-10T10:28:04.708734Z","shell.execute_reply":"2022-08-10T10:28:04.713939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot_images(df[df['label'] == 'LAA'])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:28:04.717211Z","iopub.execute_input":"2022-08-10T10:28:04.717663Z","iopub.status.idle":"2022-08-10T10:28:04.726695Z","shell.execute_reply.started":"2022-08-10T10:28:04.71762Z","shell.execute_reply":"2022-08-10T10:28:04.725426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir /kaggle/working/train_patches/","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:28:04.728955Z","iopub.execute_input":"2022-08-10T10:28:04.730157Z","iopub.status.idle":"2022-08-10T10:28:05.881794Z","shell.execute_reply.started":"2022-08-10T10:28:04.730105Z","shell.execute_reply":"2022-08-10T10:28:05.880144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clip_and_save(x):\n    image_id, img_path = x['image_id'], x['path']\n    img = pyvips.Image.thumbnail(img_path, 20000).numpy()\n    if img.shape[0] > img.shape[1]:\n        img = np.rollaxis(img, 0, 2)\n\n    img, mask = mask_median(np.array(img))\n    img = prune_image_rows_cols(img, mask)\n    img = Image.fromarray(img)\n    print('Completed: {}'.format(img_path))\n    img.save('/kaggle/working/train_patches/{}.png'.format(image_id))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:06:46.438219Z","iopub.execute_input":"2022-08-11T07:06:46.438654Z","iopub.status.idle":"2022-08-11T07:06:46.448177Z","shell.execute_reply.started":"2022-08-11T07:06:46.438619Z","shell.execute_reply":"2022-08-11T07:06:46.447229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tile_and_save(x):\n    image_id, img_path = x['image_id'], x['path']\n    img = pyvips.Image.thumbnail(img_path, 20000).numpy()\n    images = tile(img, sz=2048)\n    for idx, img in enumerate(images):\n        img = Image.fromarray(img)\n        img.save('/kaggle/working/train_patches/{}_{}.jpeg'.format(image_id, idx))\n\n    print(f'{x.name} Completed: {format(img_path)}')\n    del img, images; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:06:49.077783Z","iopub.execute_input":"2022-08-11T07:06:49.0782Z","iopub.status.idle":"2022-08-11T07:06:49.086159Z","shell.execute_reply.started":"2022-08-11T07:06:49.078167Z","shell.execute_reply":"2022-08-11T07:06:49.084676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.apply(tile_and_save, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:28:05.911284Z","iopub.execute_input":"2022-08-10T10:28:05.911687Z","iopub.status.idle":"2022-08-10T10:28:53.142818Z","shell.execute_reply.started":"2022-08-10T10:28:05.911655Z","shell.execute_reply":"2022-08-10T10:28:53.141579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**img.save('/kaggle/working/train_patches/{}_{}.jpeg'.format(image_id, idx)) **\n## Do We now need to save() jpeg after Debugs runs in /kaggle/working/train_patches.. Might be a overhead\n\n**Please remove code sections commented**","metadata":{}},{"cell_type":"code","source":"# plt.figure(figsize=[10,10])\n# imgTest = plt.imread('/kaggle/working/train_patches/006388_0_0.jpeg')\n# plt.imshow(imgTest)\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:29:57.390794Z","iopub.execute_input":"2022-08-10T10:29:57.391281Z","iopub.status.idle":"2022-08-10T10:29:58.341041Z","shell.execute_reply.started":"2022-08-10T10:29:57.391246Z","shell.execute_reply":"2022-08-10T10:29:58.33976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tile_to_hist(tile, binsArg,tileHistogram, colorCountsArray):\n    #remove white color\n    tileWhiteMask =  (((tile[:][:][0] < 255) & (tile[:][:][1] < 255) & (tile[:][:][2] < 255)) * 2) - 1\n\n    tile[:][:][0] =  tile[:][:][0] * tileWhiteMask\n    tile[:][:][1] =  tile[:][:][1] * tileWhiteMask\n    tile[:][:][2] =  tile[:][:][2] * tileWhiteMask\n\n    # Iterate throgh all the byte stream and stored the gray scale pexel count in Dataframe \n    # Even there are some meta data on this byte array assumed all as pixels because meta data are small compared to pixel data\n    for i in range(0,3):\n        hist,bins = np.histogram(tile[:,:,i], bins=binsArg)\n        tileHistogram =  np.concatenate((tileHistogram,hist))\n    colorCountsArray = np.add(colorCountsArray, tileHistogram)\n    return colorCountsArray","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:19:52.274539Z","iopub.execute_input":"2022-08-11T07:19:52.275155Z","iopub.status.idle":"2022-08-11T07:19:52.287708Z","shell.execute_reply.started":"2022-08-11T07:19:52.27511Z","shell.execute_reply":"2022-08-11T07:19:52.286607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir /kaggle/working/histogram","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:27:47.533099Z","iopub.execute_input":"2022-08-11T05:27:47.53355Z","iopub.status.idle":"2022-08-11T05:27:48.67006Z","shell.execute_reply.started":"2022-08-11T05:27:47.533513Z","shell.execute_reply":"2022-08-11T05:27:48.668069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def makeImgHistogramDataFrame(startRowIndex = 0):\n    workFilePath = '/kaggle/working'\n    trainPathesPath = f'{workFilePath}/train_patches/'\n    histogramDfColumns = [(f'R{i}') for i in range(0,256)] + [(f'G{i}') for i in range(0,256)] + [(f'B{i}') for i in range(0,256)] + ['target']\n    histogramDf = pd.DataFrame(columns=histogramDfColumns)\n    for index, row in df.iterrows():\n        if index >= startRowIndex:\n            histogramArray = np.zeros(768)\n            imgPatches = [f'{trainPathesPath}{x}' for x in os.listdir(trainPathesPath) if row['image_id'] in x]\n            for patch in imgPatches:\n                imagePatch = plt.imread(patch)\n                histogramArray = tile_to_hist(imagePatch, list(range(0,257)),np.array([]), histogramArray)\n            histogramArray = np.concatenate((histogramArray,np.array([row['target']])))\n            histogramDf.loc[len(histogramDf)] = histogramArray\n    histogramDf.to_csv('/kaggle/working/histogram/histogram.csv')\n    return histogramDf","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:06:57.858545Z","iopub.execute_input":"2022-08-11T07:06:57.859534Z","iopub.status.idle":"2022-08-11T07:06:57.870596Z","shell.execute_reply.started":"2022-08-11T07:06:57.859485Z","shell.execute_reply":"2022-08-11T07:06:57.869759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**%%time MicroBenchmarking are you Using here .Remove**","metadata":{}},{"cell_type":"code","source":"# %%time\n# hist = makeImgHistogramDataFrame(0)\n# hist","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:39:57.795694Z","iopub.execute_input":"2022-08-10T11:39:57.796077Z","iopub.status.idle":"2022-08-10T11:40:09.768888Z","shell.execute_reply.started":"2022-08-10T11:39:57.796046Z","shell.execute_reply":"2022-08-10T11:40:09.767647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"histogramDf = pd.read_csv('../input/createdhistogram/histogram.csv')\nhistogramDf.reindex()\nhistogramDf.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:25:32.074031Z","iopub.execute_input":"2022-08-11T08:25:32.07504Z","iopub.status.idle":"2022-08-11T08:25:32.288662Z","shell.execute_reply.started":"2022-08-11T08:25:32.074995Z","shell.execute_reply":"2022-08-11T08:25:32.287795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**histogramDf.drop(['target'] --> when ever a drop done, reindex() dataframe to optimize Pandas DF....histogram.head() printing not needed for Release**","metadata":{}},{"cell_type":"code","source":"target = histogramDf['target']\nhistogramDf = histogramDf.drop(['target'], axis=1)\nhistogramDf = histogramDf.loc[:, ~histogramDf.columns.str.contains('^Unnamed')]\n\nreg = LinearRegression()\nreg.fit(histogramDf,target)\n\nhistogramDf.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:25:44.373822Z","iopub.execute_input":"2022-08-11T08:25:44.37427Z","iopub.status.idle":"2022-08-11T08:25:44.665898Z","shell.execute_reply.started":"2022-08-11T08:25:44.374235Z","shell.execute_reply":"2022-08-11T08:25:44.664282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"predict_values = reg.predict(histogramDf)\n\nMSE = round(mean_squared_error(target, predict_values),2)\nRMSE = np.sqrt(MSE)\nprint(f'MSE={MSE}')\nprint(f'RMSE={RMSE}')\nprint(predict_values)\nprint(len(predict_values))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:25:48.478424Z","iopub.execute_input":"2022-08-11T08:25:48.478903Z","iopub.status.idle":"2022-08-11T08:25:48.524599Z","shell.execute_reply.started":"2022-08-11T08:25:48.478846Z","shell.execute_reply":"2022-08-11T08:25:48.523024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir /kaggle/working/test_patches\ndef tile_and_save_test_data(x):\n    image_id, img_path = x['image_id'], x['path']\n    img = pyvips.Image.thumbnail(img_path, 20000).numpy()\n    images = tile(img, sz=2048)\n    for idx, img in enumerate(images):\n        img = Image.fromarray(img)\n        img.save('/kaggle/working/test_patches/{}_{}.jpeg'.format(image_id, idx))\n\n    print(f'{x.name} Completed: {format(img_path)}')\n    del img, images; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:25:56.403265Z","iopub.execute_input":"2022-08-11T08:25:56.403713Z","iopub.status.idle":"2022-08-11T08:25:57.578962Z","shell.execute_reply.started":"2022-08-11T08:25:56.403679Z","shell.execute_reply":"2022-08-11T08:25:57.577657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfTestCsv = pd.read_csv('{}/test.csv'.format(path))\ndfTestCsv['path'] = path + 'test/' + dfTestCsv['image_id'] + '.tif'\ndfTestCsv['shape'] = dfTestCsv['path'].map(lambda x: Image.open(x).size)\ndfTestCsv[['width', 'height']] = pd.DataFrame(dfTestCsv['shape'].tolist(), columns=['width', 'height'])\ndfTestCsv['area'] = dfTestCsv['width'] * dfTestCsv['height']","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:26:02.60339Z","iopub.execute_input":"2022-08-11T08:26:02.604638Z","iopub.status.idle":"2022-08-11T08:26:02.635471Z","shell.execute_reply.started":"2022-08-11T08:26:02.604592Z","shell.execute_reply":"2022-08-11T08:26:02.634602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfTestCsv.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:26:08.064056Z","iopub.execute_input":"2022-08-11T08:26:08.064847Z","iopub.status.idle":"2022-08-11T08:26:08.08336Z","shell.execute_reply.started":"2022-08-11T08:26:08.064793Z","shell.execute_reply":"2022-08-11T08:26:08.082175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfTestCsv.apply(tile_and_save_test_data, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:26:12.262962Z","iopub.execute_input":"2022-08-11T08:26:12.264027Z","iopub.status.idle":"2022-08-11T08:27:36.245006Z","shell.execute_reply.started":"2022-08-11T08:26:12.263976Z","shell.execute_reply":"2022-08-11T08:27:36.24391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def makeImgHistogramDataFrameTest(startRowIndex = 0):\n    workFilePath = '/kaggle/working'\n    trainPathesPath = f'{workFilePath}/test_patches/'\n    histogramDfColumns = [(f'R{i}') for i in range(0,256)] + [(f'G{i}') for i in range(0,256)] + [(f'B{i}') for i in range(0,256)]\n    histogramDf = pd.DataFrame(columns=histogramDfColumns)\n    for index, row in dfTestCsv.iterrows():\n        if index >= startRowIndex:\n            histogramArray = np.zeros(768)\n            imgPatches = [f'{trainPathesPath}{x}' for x in os.listdir(trainPathesPath) if row['image_id'] in x]\n            for patch in imgPatches:\n                imagePatch = plt.imread(patch)\n                histogramArray = tile_to_hist(imagePatch, list(range(0,257)),np.array([]), histogramArray)\n            histogramDf.loc[len(histogramDf)] = histogramArray\n   # histogramDf.to_csv('/kaggle/working/histogram/histogram.csv')\n    return histogramDf","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:27:36.246702Z","iopub.execute_input":"2022-08-11T08:27:36.247061Z","iopub.status.idle":"2022-08-11T08:27:36.257457Z","shell.execute_reply.started":"2022-08-11T08:27:36.24703Z","shell.execute_reply":"2022-08-11T08:27:36.256434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testHistogramDf = makeImgHistogramDataFrameTest(0)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:27:36.2594Z","iopub.execute_input":"2022-08-11T08:27:36.260132Z","iopub.status.idle":"2022-08-11T08:28:01.916829Z","shell.execute_reply.started":"2022-08-11T08:27:36.260085Z","shell.execute_reply":"2022-08-11T08:28:01.915553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testHistogramDf.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:28:05.343253Z","iopub.execute_input":"2022-08-11T08:28:05.343678Z","iopub.status.idle":"2022-08-11T08:28:05.376169Z","shell.execute_reply.started":"2022-08-11T08:28:05.34364Z","shell.execute_reply":"2022-08-11T08:28:05.374957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_values_test = reg.predict(testHistogramDf)\npredict_values_test = np.abs(np.round(predict_values_test,2))\npredict_values_test\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:28:11.484175Z","iopub.execute_input":"2022-08-11T08:28:11.4846Z","iopub.status.idle":"2022-08-11T08:28:11.503138Z","shell.execute_reply.started":"2022-08-11T08:28:11.484564Z","shell.execute_reply":"2022-08-11T08:28:11.502069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"finalResult = pd.DataFrame(columns=np.array(['patient_id','CE','LAA']))\nfinalResult['patient_id'] = dfTestCsv['patient_id']\nfinalResult['CE'] = 1 - predict_values_test\nfinalResult['LAA'] = predict_values_test\nfinalResult","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:28:26.963534Z","iopub.execute_input":"2022-08-11T08:28:26.964004Z","iopub.status.idle":"2022-08-11T08:28:26.980781Z","shell.execute_reply.started":"2022-08-11T08:28:26.963964Z","shell.execute_reply":"2022-08-11T08:28:26.979825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir /kaggle/working/test_result\nfinalResult.to_csv('/kaggle/working/test_result/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:28:46.083919Z","iopub.execute_input":"2022-08-11T08:28:46.08434Z","iopub.status.idle":"2022-08-11T08:28:47.234153Z","shell.execute_reply.started":"2022-08-11T08:28:46.084308Z","shell.execute_reply":"2022-08-11T08:28:47.233047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}