{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n![](https://cdn.pixabay.com/photo/2022/04/08/01/01/blood-clot-7118517_960_720.png )\n# Overview \nHI,The Mayo Clinic is a nonprofit American academic medical center focused on integrated health care, education, and research. It employs over 4,500 physicians and scientists, along with another 58,400 administrative and allied health staff, across three major campuses:","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"I will explore the Data by means of:\n- Structure Investigation  \n- Quality Investigation  \n- Descriptive Statistics","metadata":{}},{"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport math\nimport os\nfrom PIL import Image\nimport cv2","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-29T12:24:23.22118Z","iopub.execute_input":"2022-08-29T12:24:23.221797Z","iopub.status.idle":"2022-08-29T12:24:23.229028Z","shell.execute_reply.started":"2022-08-29T12:24:23.22173Z","shell.execute_reply":"2022-08-29T12:24:23.227796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#lIBRARY\ndef color_channel_analysis(resized_img):\n    colors = resized_img.getbands()\n    \n    nn=len(colors)\n    plt.figure(figsize=(16,nn*10))\n    for j, channel in enumerate(resized_img.getbands()):\n        for i in range(0,3*nn,nn):\n            color = int(i/3)\n\n            # color channel\n            plt.subplot(3*nn, 3, i + 1)\n            #print(resized_img)\n            resized_img=np.array(resized_img)\n            plt.imshow(resized_img[:,:,color], aspect='auto') # align all subplots\n            plt.title(f'{colors[color]} Channel')\n\n            # histogram\n            ax = plt.subplot(3*nn, 3, i + 2)\n            df = resized_img[:,:,color].ravel()\n            ax.axvline(150,c=\"red\",linestyle=\"--\")\n            sns.histplot(df, bins=np.arange(0,255), ax=ax)\n            plt.title(f'{colors[color]} Channel Histogram')\n\n            # threshold\n            plt.subplot(3*nn, 3, i + 3) # threshold\n            thresh = resized_img.copy()\n            thresh[np.array(resized_img[:,:,color] < 150)] = 0\n            plt.imshow(thresh, aspect='auto')\n            plt.title(f'Threshold {colors[color]} Channel')\n\n    #plt.suptitle('Color Channel Analysis', y = 1.0, fontsize=16)\n    plt.tight_layout()\n    plt.show()\n    \ndef case_study(df, img_id=\"Random\"):\n    \"\"\"get image dataframe\n    Detailed explanation \n    Args:  \n        df: df\n        img_id: id\n    Returns:\n       imgdf\n    \"\"\"       \n    plt.style.use('default')\n    \n    if img_id==\"Random\":\n        img_instance = df.sample(1)\n    else:  \n        img_instance = df.loc[df.iloc[:,0]==img_id]\n    path=img_instance.path.to_string(index=False) \n    img = Image.open(path)   \n    image_resized=img.resize((300,300), Image.Resampling.LANCZOS)\n    \n    fig, ax = plt.subplots(1,2, figsize=(6,6))\n    ax[0].imshow(img)\n    ax[0].set_title('Original image')\n    \n    ax[1].imshow(image_resized)\n    ax[1].set_title('Resized image')\n    plt.tight_layout()\n    plt.show()\n    print(img_instance)\n     \n    return img, image_resized\n\ndef univar_dis(df,fea_list,flag=\"num\"):\n    \"\"\"plot distribution for a fea list in df.\n    Detailed explanation \n    Args:  \n        df: train in dataframe\n        fea_list: ['fea1','fea2']\n        flag: \n    Returns:\n       nothing\n        \n    \"\"\"      \n    #build figure, set size\n    r_n=math.ceil(len(fea_list)/3)\n    fig, ax =plt.subplots(figsize = (18, 6*r_n), nrows = r_n, ncols = 3)\n    ax=ax.reshape(-1)\n    #plot data\n    for i in range(0,len(fea_list)):\n        \n        sns.histplot(x=fea_list[i], data=df,ax=ax[i])\n        ax[i].legend() \n        ax[i].set_title(fea_list[i])\n    plt.show()\n \n\n\ndef bivar_dis(df,fea_list):\n    \"\"\"plot bivar_dis for a fea list in df.\n    Detailed explanation \n    Args:  \n        df: train in dataframe\n        fea_list: [['fea1','fea2'],['fea1','fea2'],['fea1','fea2']]\n        \n    Returns:\n       nothing\n        \n    \"\"\"   \n    #build figure, set size\n    r_n=math.ceil(len(fea_list)/3)\n    fig, ax =plt.subplots(figsize = (18, 6*r_n), nrows = r_n, ncols = 3)\n    ax=ax.reshape(-1)\n    for i,coup in enumerate(fea_list):\n        \n        #plot data\n        sns.histplot(x=coup[0],hue=coup[1], data=df,ax=ax[i])\n        ax[i].set_title(coup[0]+\" vs \"+coup[1])\n    plt.show()\n \n\n\n\ndef struct_Investigation(df,img=[]): \n    \"\"\"plot Overview Information about datatype, variable type, shape, examples\n    Detailed explanation \n    Args:  \n        df: train in dataframe\n        img: ['img1','img1'] the image feature name\n    Returns:\n       nothing\n\n    \"\"\"       \n\n    print(\"There are {} instances, {} cols in total\".format(train_df.iloc[:,0].count(),train_df.iloc[0,:].count()))\n    info = pd.DataFrame( columns=[\"variable\",\"dtype\",\"measure_scales\",\"Unique Count\",\"Null Count\",\"thumbnail\"])\n    info.variable=df.columns\n    info=info.set_index('variable')\n\n    # info.i=df.dtype\n    for fea in list(df.columns):\n        info.loc[fea,\"Null Count\"]=df[fea].isnull().sum()\n        info.loc[fea,\"dtype\"]=str(df[fea].dtypes)         \n        if str(df[fea].dtypes)=='object':\n            info.loc[fea,\"dtype\"]=list(set([type(x).__name__ for x in df[fea]]))    \n        #info.i=df.shape\n        #info.loc[fea,\"shape\"]=list(set([x.shape for x in df[fea]]))    \n\n\n        if isinstance(info.loc[fea,\"dtype\"],list):\n            print(\"Feature(s) includes more than one datatype, please do cleaning first\")\n            info.loc[fea,\"measure_scales\"]=\"Unknown\"\n\n        elif fea in img:\n            info.loc[fea,\"measure_scales\"]=\"Image_address\"\n\n        elif type(df[fea][0])==str:\n            info.loc[fea,\"thumbnail\"]=str([df[fea][0],df[fea][1],df[fea][3]])\n            if len(str(df[fea][0]).split())>1 or len(str(df[fea][2]).split())>1:\n                info.loc[fea,\"measure_scales\"]=\"Text\"\n            else:\n                info.loc[fea,\"measure_scales\"]=\"Categorical Data\"\n\n        elif np.isreal(df[fea][0]):\n            info.loc[fea,\"thumbnail\"]=str([df[fea][0],df[fea][1],df[fea][3]])\n            if df[fea].nunique()/df[fea].count()<0.1 or df[fea]<100:\n                info.loc[fea,\"measure_scales\"]=\"Categorical Data\"\n            else:\n                info.loc[fea,\"measure_scales\"]=\"numerical Data\"\n\n        else:  \n            info.loc[fea,\"measure_scales\"]=\"others\"\n\n        #info.i=df.cardinality\n        info.loc[fea,\"Unique Count\"]= str(df[fea].nunique())+ \"({p:.2f}%) of total\".format(p=df[fea].nunique()/df[fea].count()*100)\n\n    return info\n    \n    \n    \ndef get_imgdf(filepath,df2,join_cols):\n    \"\"\"get image dataframe\n    Detailed explanation \n    Args:  \n        filepath: img storage\n        df2: tabular\n        join_cols: reserved information from df2\n    Returns:\n       imgdf\n    \"\"\"       \n\n    Image.MAX_IMAGE_PIXELS = None\n    df2=df2[join_cols]\n    merg_id=join_cols[0]\n    filelist = []\n    width=[]\n    height=[]\n    imgformat=[]\n    mode=[]\n    resolution=[]\n    compression=[]\n    size=[]\n    pathl=[]\n    \n    for fname in os.listdir(filepath):\n        if not fname.endswith(\".tif\"):\n            continue\n        path = os.path.join(filepath, fname) \n        \n        img = Image.open(path)    \n        dimension = img.size                      # for getting Image Width & Height\n        img_format = img.format                            # for getting Image Format\n        img_mode = img.mode                                # for getting Image Color Mode RGB, \n        img_dpi = img.info['dpi']                   # for getting Image DPI\n        img_compr = img.info['compression']          # for getting Image Compression Mode (raw means \"no compression), tiff_lzw, group4    \n        filesize = os.path.getsize(path)                                                           # for getting Image file size (in bytes)\n        size_mb = round(os.path.getsize(path)/1024/1024, 2)                                   # Image file size (bytes to mb)\n        img_size = size_mb \n         \n        \n        #dimension.append(dimension)\n        (fname,extention)=os.path.splitext(fname)\n        pathl.append(path)\n        filelist.append(fname)\n        width.append(dimension[0])\n        height.append(dimension[1])\n        imgformat.append(img_format)\n        mode.append(img_mode)\n        resolution.append(img_dpi)\n        compression.append(img_compr)\n        size.append(img_size)\n    df=pd.DataFrame()\n    df[merg_id]=filelist\n    df['width']=width\n    df['height']=height\n    df['imgformat']=imgformat\n    df['mode']=mode\n    df['resolution']=resolution\n    df['compression']=compression\n    df['size']=size\n    df['img_aspect_ratio']=df['width']/df['height']\n    df['path']=pathl\n    df=pd.merge(df, df2, on=merg_id)\n    return df\n\n\ndef img_thumnail_with_label(image_df):\n    \"\"\"plot image thumnail\n    Detailed explanation \n    Args:  \n        image_df: img df\n \n    Returns:\n        nothing\n    \"\"\"       \n\n    Image.MAX_IMAGE_PIXELS = None  # you have to set this value to allow displaying high-resolution images\n    labellist=image_df.label.unique()\n    for label in labellist:\n        print(label)\n        CE_imgs = image_df.loc[image_df['label']==label,'path']\n        plt.style.use('default')\n        fig, axes = plt.subplots(1,3, figsize=(16,16))\n\n        for ax in axes.reshape(-1):\n            img_path = np.random.choice(CE_imgs)\n            img = Image.open(img_path)   \n            img.thumbnail((300,300), Image.Resampling.LANCZOS)\n            ax.imshow(img), ax.set_title(\"target: \"+label)\n        plt.show()\n        \n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-29T12:29:59.139811Z","iopub.execute_input":"2022-08-29T12:29:59.140269Z","iopub.status.idle":"2022-08-29T12:29:59.499703Z","shell.execute_reply.started":"2022-08-29T12:29:59.140234Z","shell.execute_reply":"2022-08-29T12:29:59.498606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load the Data ","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/mayo-clinic-strip-ai/train.csv\")\ntrain_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-29T12:24:23.281611Z","iopub.execute_input":"2022-08-29T12:24:23.282401Z","iopub.status.idle":"2022-08-29T12:24:23.312292Z","shell.execute_reply.started":"2022-08-29T12:24:23.282348Z","shell.execute_reply":"2022-08-29T12:24:23.310987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  1. Structure Investigation\n\nOverview Information about datatype, variable type, shape, examples ","metadata":{}},{"cell_type":"markdown","source":"## 1.1 Tabualr","metadata":{}},{"cell_type":"code","source":"struct_Investigation(train_df,[\"image_id\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-29T12:24:23.31569Z","iopub.execute_input":"2022-08-29T12:24:23.316831Z","iopub.status.idle":"2022-08-29T12:24:23.351336Z","shell.execute_reply.started":"2022-08-29T12:24:23.316746Z","shell.execute_reply":"2022-08-29T12:24:23.35041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.2 Img","metadata":{}},{"cell_type":"code","source":"image_df=get_imgdf('../input/mayo-clinic-strip-ai/train',train_df,[\"image_id\",\"label\"])    \nimage_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-29T12:24:23.352532Z","iopub.execute_input":"2022-08-29T12:24:23.35383Z","iopub.status.idle":"2022-08-29T12:24:25.07759Z","shell.execute_reply.started":"2022-08-29T12:24:23.353786Z","shell.execute_reply":"2022-08-29T12:24:25.076103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_thumnail_with_label(image_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-29T12:24:25.079881Z","iopub.execute_input":"2022-08-29T12:24:25.080417Z","iopub.status.idle":"2022-08-29T12:25:42.639424Z","shell.execute_reply.started":"2022-08-29T12:24:25.080382Z","shell.execute_reply":"2022-08-29T12:25:42.638019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.3 Case Study of Img","metadata":{}},{"cell_type":"code","source":"\nimg_LAA, img_resized_LAA = case_study(image_df, '08d3d8_0')\n\ncolor_channel_analysis(img_resized_LAA)","metadata":{"execution":{"iopub.status.busy":"2022-08-29T12:25:42.642649Z","iopub.execute_input":"2022-08-29T12:25:42.643199Z","iopub.status.idle":"2022-08-29T12:26:15.985628Z","shell.execute_reply.started":"2022-08-29T12:25:42.643152Z","shell.execute_reply":"2022-08-29T12:26:15.982009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Quality Investigation","metadata":{}},{"cell_type":"markdown","source":"## 2.1 Check Img Size ","metadata":{}},{"cell_type":"code","source":"univar_dis(image_df,[\"width\",\"height\",\"size\",\"img_aspect_ratio\"],flag=\"num\")  \n\ncheck=image_df[\"width\"].median()\nimage_df[\"widenarrow\"]=image_df[\"width\"].apply(lambda x: \"wide\" if x>=check else \"narrow\")\nimage_df[\"shorttall\"]=image_df[\"height\"].apply(lambda x: \"tall\" if x>=check else \"short\")\nbivar_dis(image_df,[[\"label\",\"widenarrow\"],[\"widenarrow\",\"label\"],[\"label\",\"shorttall\"],[\"shorttall\",\"label\"]]) \n \n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  2.2 Resize\n","metadata":{}},{"cell_type":"code","source":"def convert_img(img_ids):\n    for img_id in tqdm(img_ids):\n        try:\n            img_path = f'{data_path}/train/{img_id}.tif'\n            image = rasterio.open(img_path)\n            image = image.read(out_shape=(image.count, int(image_scalar), int(image_scalar)),\n                             resampling=Resampling.bilinear)\n            with rasterio.open(f'./convert_train/{img_id}.png', 'w', driver='png', height = image.shape[1], width = image.shape[2], dtype = image.dtype, count=3) as images: # count => Image Channel 개수 (RGB의 경우 3개)\n                images.write(image)\n            #image.save(f'./convert_train/{img_id}.png', 'png')\n            # Image 용량이 너무 커서 Ram 커널 반복적으로 죽는 상황 발생 => @JIRKA BOROVEC님 코드 참조 => Garbage Collection 활용, 누수되는 Ram Memory 활용\n            del image\n            gc.collect()\n            \n        except OSError as e:\n            print(e)\n            \n#convert_img(img_ids)","metadata":{"execution":{"iopub.status.busy":"2022-08-29T12:26:17.292279Z","iopub.execute_input":"2022-08-29T12:26:17.293343Z","iopub.status.idle":"2022-08-29T12:26:17.302952Z","shell.execute_reply.started":"2022-08-29T12:26:17.293295Z","shell.execute_reply":"2022-08-29T12:26:17.301656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data Cleaning, Filtering  etc,.","metadata":{}},{"cell_type":"markdown","source":"# 3. Descriptive Statistic","metadata":{}},{"cell_type":"markdown","source":" ## 3.1 Univariate\n    \nMeasures of distribution,Central Tendency, and Variation or spread of indivisual variable","metadata":{}},{"cell_type":"code","source":"# distribution of center_id， image_num， label\n#test\nunivar_dis(train_df,[\"center_id\",\"image_num\",\"label\",\"image_num\",\"label\",\"image_num\",\"label\"],flag=\"num\") \n","metadata":{"execution":{"iopub.status.busy":"2022-08-29T12:29:23.776637Z","iopub.execute_input":"2022-08-29T12:29:23.777096Z","iopub.status.idle":"2022-08-29T12:29:25.186315Z","shell.execute_reply.started":"2022-08-29T12:29:23.777062Z","shell.execute_reply":"2022-08-29T12:29:25.185137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" ## 3.1 Bivariate\n    \nMostly Focus on Patterns and Covariation between the variables, relationship, compare","metadata":{}},{"cell_type":"markdown","source":"### 3.1.1 Bar Chart(Hue)\n\n\nCorrelation between every two nominal data SHOWN in attribute space(three pattern)","metadata":{}},{"cell_type":"code","source":"\nbivar_dis(train_df,[[\"label\",\"center_id\"],[\"center_id\",\"label\"],[\"label\",\"image_num\"],[\"center_id\",\"image_num\",]])   ","metadata":{"execution":{"iopub.status.busy":"2022-08-29T12:30:04.582452Z","iopub.execute_input":"2022-08-29T12:30:04.582926Z","iopub.status.idle":"2022-08-29T12:30:06.048798Z","shell.execute_reply.started":"2022-08-29T12:30:04.582888Z","shell.execute_reply":"2022-08-29T12:30:06.047269Z"},"trusted":true},"execution_count":null,"outputs":[]}]}