{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#BASIC\nimport numpy as np \nimport pandas as pd \nimport os\n\n# DATA visualization\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport PIL\nfrom IPython.display import Image, display\nfrom plotly import graph_objs as go\nimport plotly.express as px\nimport plotly.figure_factory as ff\n\nimport openslide","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:22.042782Z","iopub.execute_input":"2023-04-01T10:17:22.043278Z","iopub.status.idle":"2023-04-01T10:17:22.051516Z","shell.execute_reply.started":"2023-04-01T10:17:22.043241Z","shell.execute_reply":"2023-04-01T10:17:22.050018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_FOLDER = \"/kaggle/input/prostate-cancer-grade-assessment/\"\n!ls {BASE_FOLDER}","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:22.066407Z","iopub.execute_input":"2023-04-01T10:17:22.067447Z","iopub.status.idle":"2023-04-01T10:17:23.170291Z","shell.execute_reply.started":"2023-04-01T10:17:22.067391Z","shell.execute_reply":"2023-04-01T10:17:23.168919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask_dir = f'{BASE_FOLDER}/train_label_masks'","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:23.173025Z","iopub.execute_input":"2023-04-01T10:17:23.173671Z","iopub.status.idle":"2023-04-01T10:17:23.180567Z","shell.execute_reply.started":"2023-04-01T10:17:23.173618Z","shell.execute_reply":"2023-04-01T10:17:23.179056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(BASE_FOLDER+\"train.csv\")\ntest = pd.read_csv(BASE_FOLDER+\"test.csv\")\nsub = pd.read_csv(BASE_FOLDER+\"sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:23.182024Z","iopub.execute_input":"2023-04-01T10:17:23.1824Z","iopub.status.idle":"2023-04-01T10:17:23.222052Z","shell.execute_reply.started":"2023-04-01T10:17:23.182367Z","shell.execute_reply":"2023-04-01T10:17:23.220841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:23.224066Z","iopub.execute_input":"2023-04-01T10:17:23.22463Z","iopub.status.idle":"2023-04-01T10:17:23.251463Z","shell.execute_reply.started":"2023-04-01T10:17:23.224578Z","shell.execute_reply":"2023-04-01T10:17:23.249919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop([7273],inplace=True) #Mislabelled","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:23.256969Z","iopub.execute_input":"2023-04-01T10:17:23.257379Z","iopub.status.idle":"2023-04-01T10:17:23.268145Z","shell.execute_reply.started":"2023-04-01T10:17:23.257343Z","shell.execute_reply":"2023-04-01T10:17:23.266721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['gleason_score'] = train['gleason_score'].apply(lambda x: \"0+0\" if x==\"negative\" else x)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:23.269604Z","iopub.execute_input":"2023-04-01T10:17:23.269963Z","iopub.status.idle":"2023-04-01T10:17:23.285635Z","shell.execute_reply.started":"2023-04-01T10:17:23.269932Z","shell.execute_reply":"2023-04-01T10:17:23.284436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = train.groupby('isup_grade').count()['image_id'].reset_index().sort_values(by='image_id',ascending=False)\ntemp.style.background_gradient(cmap='Purples')","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:23.287321Z","iopub.execute_input":"2023-04-01T10:17:23.288018Z","iopub.status.idle":"2023-04-01T10:17:23.385111Z","shell.execute_reply.started":"2023-04-01T10:17:23.287979Z","shell.execute_reply":"2023-04-01T10:17:23.383964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure(go.Funnelarea(\n    text =temp.isup_grade,\n    values = temp.image_id,\n    title = {\"position\": \"top center\", \"text\": \"Funnel-Chart of ISUP_grade Distribution\"}\n    ))\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:23.386323Z","iopub.execute_input":"2023-04-01T10:17:23.386735Z","iopub.status.idle":"2023-04-01T10:17:23.478404Z","shell.execute_reply.started":"2023-04-01T10:17:23.386701Z","shell.execute_reply":"2023-04-01T10:17:23.477249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(temp, x='isup_grade', y='image_id',\n             hover_data=['image_id', 'isup_grade'], color='image_id',\n             labels={'pop':'population of Canada'}, height=400)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:23.480075Z","iopub.execute_input":"2023-04-01T10:17:23.480556Z","iopub.status.idle":"2023-04-01T10:17:24.735725Z","shell.execute_reply.started":"2023-04-01T10:17:23.480521Z","shell.execute_reply":"2023-04-01T10:17:24.734408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(10,6))\nax = sns.countplot(x=\"isup_grade\", hue=\"data_provider\", data=train)\nfor p in ax.patches:\n    height = p.get_height()\n    ax.text(p.get_x()+p.get_width()/2,\n                height +3,\n                '{:1.2f}%'.format(100*height/10616),\n                ha=\"center\")","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:24.737555Z","iopub.execute_input":"2023-04-01T10:17:24.740272Z","iopub.status.idle":"2023-04-01T10:17:25.143566Z","shell.execute_reply.started":"2023-04-01T10:17:24.740213Z","shell.execute_reply":"2023-04-01T10:17:25.142385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = train.groupby('gleason_score').count()['image_id'].reset_index().sort_values(by='image_id',ascending=False)\ntemp.style.background_gradient(cmap='Reds')","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:25.144985Z","iopub.execute_input":"2023-04-01T10:17:25.145399Z","iopub.status.idle":"2023-04-01T10:17:25.170101Z","shell.execute_reply.started":"2023-04-01T10:17:25.145366Z","shell.execute_reply":"2023-04-01T10:17:25.168809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure(go.Funnelarea(\n    text =temp.gleason_score,\n    values = temp.image_id,\n    title = {\"position\": \"top center\", \"text\": \"Funnel-Chart of ISUP_grade Distribution\"}\n    ))\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:25.171422Z","iopub.execute_input":"2023-04-01T10:17:25.171807Z","iopub.status.idle":"2023-04-01T10:17:25.187838Z","shell.execute_reply.started":"2023-04-01T10:17:25.171775Z","shell.execute_reply":"2023-04-01T10:17:25.185712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(temp, x='gleason_score', y='image_id',\n             hover_data=['image_id', 'gleason_score'], color='image_id',\n             labels={'pop':'population of Canada'}, height=400)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:25.189711Z","iopub.execute_input":"2023-04-01T10:17:25.190479Z","iopub.status.idle":"2023-04-01T10:17:25.279396Z","shell.execute_reply.started":"2023-04-01T10:17:25.190412Z","shell.execute_reply":"2023-04-01T10:17:25.277995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nVisualizing the GLEASON_SCORE distribution wrt Data_providers\n'''\n\nfig = plt.figure(figsize=(10,6))\nax = sns.countplot(x=\"gleason_score\", hue=\"data_provider\", data=train)\nfor p in ax.patches:\n    height = p.get_height()\n    ax.text(p.get_x()+p.get_width()/2.,\n                height + 3,\n                '{:1.2f}%'.format(100*height/10616),\n                ha=\"center\")","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:25.285816Z","iopub.execute_input":"2023-04-01T10:17:25.286422Z","iopub.status.idle":"2023-04-01T10:17:25.698457Z","shell.execute_reply.started":"2023-04-01T10:17:25.286357Z","shell.execute_reply":"2023-04-01T10:17:25.697535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Open the image (does not yet read the image into memory)\nexample = openslide.OpenSlide(os.path.join(BASE_FOLDER+\"train_images\", '005e66f06bce9c2e49142536caf2f6ee.tiff'))\n\n# Read a specific region of the image starting at upper left coordinate (x=17800, y=19500) on level 0 and extracting a 256*256 pixel patch.\n# At this point image data is read from the file and loaded into memory.\npatch = example.read_region((17800,19500), 0, (256, 256))\n\n# Display the image\ndisplay(patch)\n\n# Close the opened slide after use\nexample.close()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:25.699798Z","iopub.execute_input":"2023-04-01T10:17:25.700847Z","iopub.status.idle":"2023-04-01T10:17:25.862734Z","shell.execute_reply.started":"2023-04-01T10:17:25.700809Z","shell.execute_reply":"2023-04-01T10:17:25.861453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.set_index('image_id')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:25.864066Z","iopub.execute_input":"2023-04-01T10:17:25.864506Z","iopub.status.idle":"2023-04-01T10:17:25.878571Z","shell.execute_reply.started":"2023-04-01T10:17:25.864468Z","shell.execute_reply":"2023-04-01T10:17:25.877655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_values(image,max_size=(600,400)):\n    slide = openslide.OpenSlide(os.path.join(BASE_FOLDER+\"train_images\", f'{image}.tiff'))\n    \n    # Here we compute the \"pixel spacing\": the physical size of a pixel in the image.\n    # OpenSlide gives the resolution in centimeters so we convert this to microns.\n    f,ax =  plt.subplots(2 ,figsize=(6,16))\n    spacing = 1 / (float(slide.properties['tiff.XResolution']) / 10000)\n    patch = slide.read_region((1780,1950), 0, (256, 256)) #ZOOMED FUGURE\n    ax[0].imshow(patch) \n    ax[0].set_title('Zoomed Image')\n    ax[1].imshow(slide.get_thumbnail(size=max_size)) #UNZOOMED FIGURE\n    ax[1].set_title('Full Image')\n    \n    \n    print(f\"File id: {slide}\")\n    print(f\"Dimensions: {slide.dimensions}\")\n    print(f\"Microns per pixel / pixel spacing: {spacing:.3f}\")\n    print(f\"Number of levels in the image: {slide.level_count}\")\n    print(f\"Downsample factor per level: {slide.level_downsamples}\")\n    print(f\"Dimensions of levels: {slide.level_dimensions}\\n\\n\")\n    print(f\"ISUP grade: {train.loc[image, 'isup_grade']}\")\n    print(f\"Gleason score: {train.loc[image, 'gleason_score']}\")","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:25.879538Z","iopub.execute_input":"2023-04-01T10:17:25.879877Z","iopub.status.idle":"2023-04-01T10:17:25.892414Z","shell.execute_reply.started":"2023-04-01T10:17:25.879847Z","shell.execute_reply":"2023-04-01T10:17:25.89129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Segmentation","metadata":{}},{"cell_type":"code","source":"get_values('07a7ef0ba3bb0d6564a73f4f3e1c2293')","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:25.893691Z","iopub.execute_input":"2023-04-01T10:17:25.894179Z","iopub.status.idle":"2023-04-01T10:17:26.788962Z","shell.execute_reply.started":"2023-04-01T10:17:25.894124Z","shell.execute_reply":"2023-04-01T10:17:26.787663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_images(images):\n    '''\n    This function takes in input a list of images. It then iterates through the image making openslide objects , on which different functions\n    for getting out information can be called later\n    '''\n    f, ax = plt.subplots(5,3, figsize=(18,22))\n    for i, image in enumerate(images):\n        slide = openslide.OpenSlide(os.path.join(BASE_FOLDER+\"train_images\", f'{image}.tiff'))\n        # Making Openslide Object\n        #Here we compute the \"pixel spacing\": the physical size of a pixel in the image,\n        #OpenSlide gives the resolution in centimeters so we convert this to microns\n        spacing = 1/(float(slide.properties['tiff.XResolution']) / 10000)\n        patch = slide.read_region((1780,1950), 0, (256, 256)) #Reading the image as before betweeen x=1780 to y=1950 and of pixel size =256*256\n        ax[i//3, i%3].imshow(patch) #Displaying Image\n        slide.close()       \n        ax[i//3, i%3].axis('off')\n        image_id = image\n        data_provider = train.loc[image, 'data_provider']\n        isup_grade = train.loc[image, 'isup_grade']\n        gleason_score = train.loc[image, 'gleason_score']\n        ax[i//3, i%3].set_title(f\"ID: {image_id}\\nSource: {data_provider} ISUP: {isup_grade} Gleason: {gleason_score}\")\n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:26.790645Z","iopub.execute_input":"2023-04-01T10:17:26.791127Z","iopub.status.idle":"2023-04-01T10:17:26.800839Z","shell.execute_reply.started":"2023-04-01T10:17:26.791092Z","shell.execute_reply":"2023-04-01T10:17:26.799864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images = [\n'07a7ef0ba3bb0d6564a73f4f3e1c2293',\n    '037504061b9fba71ef6e24c48c6df44d',\n    '035b1edd3d1aeeffc77ce5d248a01a53',\n    '059cbf902c5e42972587c8d17d49efed',\n    '06a0cbd8fd6320ef1aa6f19342af2e68',\n    '06eda4a6faca84e84a781fee2d5f47e1',\n    '0a4b7a7499ed55c71033cefb0765e93d',\n    '0838c82917cd9af681df249264d2769c',\n    '046b35ae95374bfb48cdca8d7c83233f',\n    '074c3e01525681a275a42282cd21cbde',\n    '05abe25c883d508ecc15b6e857e59f32',\n    '05f4e9415af9fdabc19109c980daf5ad',\n    '060121a06476ef401d8a21d6567dee6d',\n    '068b0e3be4c35ea983f77accf8351cc8',\n    '08f055372c7b8a7e1df97c6586542ac8'\n]\ndisplay_images(images)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:26.802237Z","iopub.execute_input":"2023-04-01T10:17:26.802937Z","iopub.status.idle":"2023-04-01T10:17:29.536935Z","shell.execute_reply.started":"2023-04-01T10:17:26.802901Z","shell.execute_reply":"2023-04-01T10:17:29.535722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\ndata=pd.read_csv('../input/prostate-cancer/Prostate_Cancer.csv')\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:29.538831Z","iopub.execute_input":"2023-04-01T10:17:29.539957Z","iopub.status.idle":"2023-04-01T10:17:29.577652Z","shell.execute_reply.started":"2023-04-01T10:17:29.539914Z","shell.execute_reply":"2023-04-01T10:17:29.576385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:29.579379Z","iopub.execute_input":"2023-04-01T10:17:29.579843Z","iopub.status.idle":"2023-04-01T10:17:29.599079Z","shell.execute_reply.started":"2023-04-01T10:17:29.579807Z","shell.execute_reply":"2023-04-01T10:17:29.597775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=data.drop_duplicates()\ndata.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:29.600933Z","iopub.execute_input":"2023-04-01T10:17:29.602278Z","iopub.status.idle":"2023-04-01T10:17:29.626718Z","shell.execute_reply.started":"2023-04-01T10:17:29.602228Z","shell.execute_reply":"2023-04-01T10:17:29.625413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nenc=LabelEncoder()\ndata['diagnosis_result']=enc.fit_transform(data['diagnosis_result'])\ndata.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:29.628778Z","iopub.execute_input":"2023-04-01T10:17:29.629522Z","iopub.status.idle":"2023-04-01T10:17:29.728488Z","shell.execute_reply.started":"2023-04-01T10:17:29.629481Z","shell.execute_reply":"2023-04-01T10:17:29.727247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.describe()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:29.729976Z","iopub.execute_input":"2023-04-01T10:17:29.730348Z","iopub.status.idle":"2023-04-01T10:17:29.776628Z","shell.execute_reply.started":"2023-04-01T10:17:29.730315Z","shell.execute_reply":"2023-04-01T10:17:29.775302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for column in data.columns:\n    data[column] = (data[column] - data[column].min()) / (data[column].max() - data[column].min()) \ndata.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:29.778052Z","iopub.execute_input":"2023-04-01T10:17:29.778439Z","iopub.status.idle":"2023-04-01T10:17:29.807199Z","shell.execute_reply.started":"2023-04-01T10:17:29.77839Z","shell.execute_reply":"2023-04-01T10:17:29.805862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:29.811043Z","iopub.execute_input":"2023-04-01T10:17:29.811455Z","iopub.status.idle":"2023-04-01T10:17:29.822359Z","shell.execute_reply.started":"2023-04-01T10:17:29.811408Z","shell.execute_reply":"2023-04-01T10:17:29.82136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=data.drop(['id'],axis=1)\ndata.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:29.823668Z","iopub.execute_input":"2023-04-01T10:17:29.824338Z","iopub.status.idle":"2023-04-01T10:17:29.842384Z","shell.execute_reply.started":"2023-04-01T10:17:29.824302Z","shell.execute_reply":"2023-04-01T10:17:29.840848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['diagnosis_result'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:29.844613Z","iopub.execute_input":"2023-04-01T10:17:29.845174Z","iopub.status.idle":"2023-04-01T10:17:29.85606Z","shell.execute_reply.started":"2023-04-01T10:17:29.845125Z","shell.execute_reply":"2023-04-01T10:17:29.85511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cls_0=data[data['diagnosis_result']==0]\ncls_1=data[data['diagnosis_result']==1]","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:29.858282Z","iopub.execute_input":"2023-04-01T10:17:29.858672Z","iopub.status.idle":"2023-04-01T10:17:29.870066Z","shell.execute_reply.started":"2023-04-01T10:17:29.858617Z","shell.execute_reply":"2023-04-01T10:17:29.869043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_class_1_over = cls_1.sample(250, replace=True)\ndf_class_0_over = cls_0.sample(250, replace=True)\ndf_test_over = pd.concat([df_class_0_over, df_class_1_over], axis=0)\ndf_test_over.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:29.871357Z","iopub.execute_input":"2023-04-01T10:17:29.871986Z","iopub.status.idle":"2023-04-01T10:17:29.892303Z","shell.execute_reply.started":"2023-04-01T10:17:29.871948Z","shell.execute_reply":"2023-04-01T10:17:29.891266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y1=df_test_over['diagnosis_result']\ndf_test_over=df_test_over.drop(['diagnosis_result'],axis=1)\nX1=df_test_over","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:29.893695Z","iopub.execute_input":"2023-04-01T10:17:29.89452Z","iopub.status.idle":"2023-04-01T10:17:29.90624Z","shell.execute_reply.started":"2023-04-01T10:17:29.894476Z","shell.execute_reply":"2023-04-01T10:17:29.905287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX1_s_train,X1_s_test ,y1_s_train, y1_s_test = train_test_split(X1,y1,\n                                                   test_size=0.25,\n                                                   random_state=0,\n                                                  shuffle = True,\n                                                  stratify = y1)\n\nprint('training data shape is :{}.'.format(X1_s_train.shape))\nprint('training label shape is :{}.'.format(y1_s_train.shape))\nprint('testing data shape is :{}.'.format(X1_s_test.shape))\nprint('testing label shape is :{}.'.format(y1_s_test.shape))","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:29.907403Z","iopub.execute_input":"2023-04-01T10:17:29.908324Z","iopub.status.idle":"2023-04-01T10:17:29.989748Z","shell.execute_reply.started":"2023-04-01T10:17:29.908285Z","shell.execute_reply":"2023-04-01T10:17:29.98851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.svm import SVC\nsvc_s_model = SVC(kernel='poly',gamma=8)\nsvc_s_model.fit(X1_s_train, y1_s_train)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:29.991329Z","iopub.execute_input":"2023-04-01T10:17:29.992509Z","iopub.status.idle":"2023-04-01T10:17:30.123337Z","shell.execute_reply.started":"2023-04-01T10:17:29.992459Z","shell.execute_reply":"2023-04-01T10:17:30.121994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBClassifier\nxgb=XGBClassifier()\nxgb.fit(X1_s_train,y1_s_train)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:30.124896Z","iopub.execute_input":"2023-04-01T10:17:30.125278Z","iopub.status.idle":"2023-04-01T10:17:30.534825Z","shell.execute_reply.started":"2023-04-01T10:17:30.125236Z","shell.execute_reply":"2023-04-01T10:17:30.5338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing Necessary Libraries\n# Displaying the sample image - Monochrome Format\nfrom skimage import data\nfrom skimage import filters\nfrom skimage.color import rgb2gray\nimport matplotlib.pyplot as plt\n\nfrom torchvision.transforms import ToTensor\nimport rasterio\nfrom rasterio.plot import show\nimport numpy as np\n\npath = \"/kaggle/input/prostate-cancer-grade-assessment/train_images/07a7ef0ba3bb0d6564a73f4f3e1c2293.tiff\"\n\nfrom PIL import Image\nImage.MAX_IMAGE_PIXELS = None\npil_image = Image.open(path)\n\ngray_coffee = rgb2gray(pil_image)\n\n# Setting the plot size to 15,15\nplt.figure(figsize=(15, 15))\n\nfor i in range(10):\n    # Iterating different thresholds\n    binarized_gray = (gray_coffee > i*0.1)*1\n    plt.subplot(5,2,i+1)\n\n    # Rounding of the threshold\n    # value to 1 decimal point\n    plt.title(\"Threshold: >\"+str(round(i*0.1,1)))\n\n    # Displaying the binarized image\n    # of various thresholds\n    plt.imshow(binarized_gray, cmap = 'gray')\n\n    plt.tight_layout()\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-04-01T11:13:41.122251Z","iopub.execute_input":"2023-04-01T11:13:41.123135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, confusion_matrix, classification_report\npredictions= svc_s_model.predict(X1_s_train)\npercentage=svc_s_model.score(X1_s_train,y1_s_train)\nres=confusion_matrix(y1_s_train,predictions)\nprint(\"Training confusion matrix\")\nprint(res)\npredictions= svc_s_model.predict(X1_s_test)\npercentage=svc_s_model.score(X1_s_test,y1_s_test)\nres=confusion_matrix(y1_s_test,predictions)\nprint(\"validation confusion matrix\")\nprint(res)\nprint(classification_report(y1_s_test, predictions))\n# check the accuracy on the training set\nprint('training accuracy = '+str(svc_s_model.score(X1_s_train, y1_s_train)*100))\nprint('testing accuracy = '+str(svc_s_model.score(X1_s_test, y1_s_test)*100))","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:30.539168Z","iopub.execute_input":"2023-04-01T10:17:30.540756Z","iopub.status.idle":"2023-04-01T10:17:30.578651Z","shell.execute_reply.started":"2023-04-01T10:17:30.540706Z","shell.execute_reply":"2023-04-01T10:17:30.577469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, confusion_matrix, classification_report\npredictions= xgb.predict(X1_s_train)\npercentage=xgb.score(X1_s_train,y1_s_train)\nres=confusion_matrix(y1_s_train,predictions)\nprint(\"Training confusion matrix\")\nprint(res)\npredictions= xgb.predict(X1_s_test)\npercentage=xgb.score(X1_s_test,y1_s_test)\nres=confusion_matrix(y1_s_test,predictions)\nprint(\"validation confusion matrix\")\nprint(res)\nprint(classification_report(y1_s_test, predictions))\n# check the accuracy on the training set\nprint('training accuracy = '+str(xgb.score(X1_s_train, y1_s_train)*100))\nprint('testing accuracy = '+str(xgb.score(X1_s_test, y1_s_test)*100))","metadata":{"execution":{"iopub.status.busy":"2023-04-01T10:17:30.580219Z","iopub.execute_input":"2023-04-01T10:17:30.580623Z","iopub.status.idle":"2023-04-01T10:17:30.636502Z","shell.execute_reply.started":"2023-04-01T10:17:30.580588Z","shell.execute_reply":"2023-04-01T10:17:30.63551Z"},"trusted":true},"execution_count":null,"outputs":[]}]}