{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport os\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-02T10:58:34.495642Z","iopub.execute_input":"2022-10-02T10:58:34.496716Z","iopub.status.idle":"2022-10-02T10:58:35.537833Z","shell.execute_reply.started":"2022-10-02T10:58:34.496613Z","shell.execute_reply":"2022-10-02T10:58:35.53668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reading in the train data\ntrain_csv = pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')\nprint(train_csv.shape)\ndisplay(train_csv.head())","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:58:35.539622Z","iopub.execute_input":"2022-10-02T10:58:35.539958Z","iopub.status.idle":"2022-10-02T10:58:35.571113Z","shell.execute_reply.started":"2022-10-02T10:58:35.539926Z","shell.execute_reply":"2022-10-02T10:58:35.570044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reading in the test data\ntest_csv = pd.read_csv('../input/mayo-clinic-strip-ai/test.csv')\nprint(test_csv.shape)\ndisplay(test_csv)\n\n# Initial euphoria at seeing only 4 rows dissipated almost immediately, on realizing this is basically a placeholder.\n# A hill-climbing opportunity washed away unceremoniously.","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:58:35.572364Z","iopub.execute_input":"2022-10-02T10:58:35.572696Z","iopub.status.idle":"2022-10-02T10:58:35.593538Z","shell.execute_reply.started":"2022-10-02T10:58:35.572658Z","shell.execute_reply":"2022-10-02T10:58:35.590458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CURIOUS - Class balance between CE and LAA. Maybe I should just think of the heart as a very large artery.\n#           Best not digress though, that will not help my cause in any way.\n\nsns.countplot(x=train_csv.label);\ndf = pd.DataFrame(train_csv.label.value_counts())\ndf['percentage'] = df.label.apply(lambda x: 100*x/df.label.sum()).round(3)\nce_percent = df.loc['CE','percentage'] # I am going to use this for the dumb-model\ndisplay(df)\nprint(f'Percentage of clots with CE etiology: {ce_percent} %')","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:58:35.596067Z","iopub.execute_input":"2022-10-02T10:58:35.596419Z","iopub.status.idle":"2022-10-02T10:58:35.797143Z","shell.execute_reply.started":"2022-10-02T10:58:35.596387Z","shell.execute_reply":"2022-10-02T10:58:35.796223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CURIOUS As we go row by row, is there a point when the CE and LAA ratio approximates 1:1?\nCE_count=0\nLAA_count=0\nfor idx in train_csv.index:\n    if train_csv.loc[idx,'label']=='CE': CE_count+=1\n    else: LAA_count+=1\n    if idx==0: continue    \n    print(f'IDX {idx}\\tLAA:{100*LAA_count/(idx+1):.0f} %\\tCE:{100*CE_count/(idx+1):.0f} %')\n    \n# Nope, after the first few, the proportion is reasonably well maintained.    ","metadata":{"execution":{"iopub.status.busy":"2022-10-02T11:05:08.942421Z","iopub.execute_input":"2022-10-02T11:05:08.942845Z","iopub.status.idle":"2022-10-02T11:05:08.970564Z","shell.execute_reply.started":"2022-10-02T11:05:08.94281Z","shell.execute_reply":"2022-10-02T11:05:08.969173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CURIOUS How many patients in train data have more than one image to their name?\nmto = len(train_csv[train_csv.image_num > 0])\nprint(f'{mto} patients have more than one image to their name - {100*mto/len(train_csv):.2f} %')\ndel mto","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:58:35.798684Z","iopub.execute_input":"2022-10-02T10:58:35.799039Z","iopub.status.idle":"2022-10-02T10:58:35.806952Z","shell.execute_reply.started":"2022-10-02T10:58:35.799005Z","shell.execute_reply":"2022-10-02T10:58:35.80581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CURIOUS What is the people / number of images distribution?\n\nprint(f'The maximum number of images for a person is {train_csv.image_num.max()+1}\\n')\n\nfours  = len(train_csv[train_csv.image_num==4])         # label of 4 means this person has 5 images in there\nthrees = len(train_csv[train_csv.image_num==3]) - fours # since those with 5 images will also have images labelled [0,1,2,3], so let's not recount those guys\ntwos   = len(train_csv[train_csv.image_num==2]) - fours - threes\nones   = len(train_csv[train_csv.image_num==1]) - fours - threes - twos\nzeros  = len(train_csv[train_csv.image_num==0]) - fours - threes - twos - ones\n\n# Or maybe a simpler way to get the same numbers - sigh !\n# train_csv.groupby(\"patient_id\").image_num.size().value_counts()\n\n# display nicely !!\nindex  = ['1 image']\nindex.extend([str(x)+' images' for x in range(2,6)])\nkount  = [zeros, ones, twos, threes, fours] \n# print(sum(kount), train_csv.patient_id.nunique()) ## just a sanity check\ndf = pd.DataFrame(data=kount, index=index, columns=['Number of persons']).T\ndisplay(df) \ndf.plot.bar(rot=0);\n\n# Cleanup\ndel zeros, ones, twos, threes, fours, kount, index, df ","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:58:35.808904Z","iopub.execute_input":"2022-10-02T10:58:35.809705Z","iopub.status.idle":"2022-10-02T10:58:36.047591Z","shell.execute_reply.started":"2022-10-02T10:58:35.809648Z","shell.execute_reply":"2022-10-02T10:58:36.04643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CONJECTURE I know there must be that odd patient who has been labelled both CE and LAA - the \"CE-LAA\" variant.\n#            Hang in there, I shall find you.\n\nodd_patients = train_csv.groupby(\"patient_id\").label.nunique()\nprint(list(odd_patients[odd_patients==2]))\ndel odd_patients\n\n# A lonely list - hmm, the metadata looks clean. Oh well.","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:58:36.049321Z","iopub.execute_input":"2022-10-02T10:58:36.05002Z","iopub.status.idle":"2022-10-02T10:58:36.059909Z","shell.execute_reply.started":"2022-10-02T10:58:36.049959Z","shell.execute_reply":"2022-10-02T10:58:36.058802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CURIOUS How many centres are we talking about here, and how many patients at each of those?\nprint(f'There are {train_csv.center_id.nunique()} different centres')\nprint()\npd.DataFrame(train_csv.groupby(by='center_id').patient_id.count().sort_values(ascending=False))\n\n# Does it matter? I guess we'll find out later. Maybe one centre is only getting CE","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:58:36.061467Z","iopub.execute_input":"2022-10-02T10:58:36.062175Z","iopub.status.idle":"2022-10-02T10:58:36.079624Z","shell.execute_reply.started":"2022-10-02T10:58:36.062109Z","shell.execute_reply":"2022-10-02T10:58:36.078242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Curious - Is some centre only getting CE (or only LAA, for that matter?)\n\ndf = pd.DataFrame(train_csv.groupby(by='center_id').label.value_counts())\ndf['patient_count'] = 0\nfor centre in range(1,12):\n    df.loc[centre,'patient_count'] = df.loc[centre,'label'].sum()    \ndf['Percentage'] = (100*df.label/df.patient_count).round(2)\ndisplay(df.sort_values(by='Percentage', ascending=False).head(11)) # Only need to see one statistic per centre\n\ndel df, centre\n\n# Centre 3 seems a bit mixed up, but nothing extraordinary here, overall, I guess.","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:58:36.081171Z","iopub.execute_input":"2022-10-02T10:58:36.082564Z","iopub.status.idle":"2022-10-02T10:58:36.115029Z","shell.execute_reply.started":"2022-10-02T10:58:36.082529Z","shell.execute_reply":"2022-10-02T10:58:36.114194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Time for the images !!\n\n# HOUSEKEEPING I need all paths into my dataframes\n\ntrain_csv['image_path'] = train_csv.image_id.apply(lambda x: os.path.join(\"../input/mayo-clinic-strip-ai/train/\", x+\".tif\"))\ntest_csv['image_path']  = test_csv.image_id.apply(lambda x: os.path.join(\"../input/mayo-clinic-strip-ai/test/\", x+\".tif\"))\n\ndisplay(train_csv.head())\ndisplay(test_csv)\n\n# Looks alright.","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:58:36.118116Z","iopub.execute_input":"2022-10-02T10:58:36.11862Z","iopub.status.idle":"2022-10-02T10:58:36.142089Z","shell.execute_reply.started":"2022-10-02T10:58:36.118589Z","shell.execute_reply":"2022-10-02T10:58:36.140844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from skimage import io","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:58:36.143366Z","iopub.execute_input":"2022-10-02T10:58:36.143698Z","iopub.status.idle":"2022-10-02T10:58:36.729207Z","shell.execute_reply.started":"2022-10-02T10:58:36.143659Z","shell.execute_reply":"2022-10-02T10:58:36.727795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_path = train_csv['image_path'][1]    # number 0 is just too big an image\nimg = io.imread(image_path)\nimg.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:58:36.730544Z","iopub.execute_input":"2022-10-02T10:58:36.731358Z","iopub.status.idle":"2022-10-02T10:58:40.408013Z","shell.execute_reply.started":"2022-10-02T10:58:36.731318Z","shell.execute_reply":"2022-10-02T10:58:40.407087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(img)\nplt.show()\ndel img\n# Am I looking at the same thing in two different corners?","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:58:40.409656Z","iopub.execute_input":"2022-10-02T10:58:40.410367Z","iopub.status.idle":"2022-10-02T10:58:54.334061Z","shell.execute_reply.started":"2022-10-02T10:58:40.410319Z","shell.execute_reply":"2022-10-02T10:58:54.332396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's try another one\nimage_path = train_csv['image_path'][4] # image number 2 is too big as well, it seems\nimg = io.imread(image_path)\nplt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:58:54.335592Z","iopub.execute_input":"2022-10-02T10:58:54.335976Z","iopub.status.idle":"2022-10-02T10:59:10.091693Z","shell.execute_reply.started":"2022-10-02T10:58:54.33594Z","shell.execute_reply":"2022-10-02T10:59:10.090273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I think I need size information on these huge billboards of images\n# Judging by the two images I just saw, even the aspect ratios are all over the place. Hmm...\n\n\nfrom PIL import Image\nImage.MAX_IMAGE_PIXELS = None\n\ntrain_csv[\"image_size\"]   = train_csv.image_path.apply(lambda x: Image.open(x).size)\ntrain_csv[\"image_width\"]  = train_csv[\"image_size\"].apply(lambda x: int(x[0]))\ntrain_csv[\"image_height\"] = train_csv[\"image_size\"].apply(lambda x: int(x[1]))\ntrain_csv[\"aspect_ratio\"] = train_csv[\"image_width\"]/train_csv[\"image_height\"]\n\ndel train_csv[\"image_size\"]\n\nprint(f'Widest image is a whopping {train_csv[\"image_width\"].max()} pixels')\nprint(f'Narrowest image is a mere {train_csv[\"image_width\"].min()} pixels')\nprint()\nprint(f'Tallest image is a looming {train_csv[\"image_height\"].max()} pixels')\nprint(f'Shortest image is an elfin {train_csv[\"image_height\"].min()} pixels')\nprint()\nprint(f'Aspect ratios - vertical elongation can go as bad as {1/train_csv[\"aspect_ratio\"].min()}')\nprint(f'Aspect ratios - horizontal flatness can go as bad as {train_csv[\"aspect_ratio\"].max()}')\n\ntrain_csv.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:59:10.093971Z","iopub.execute_input":"2022-10-02T10:59:10.094412Z","iopub.status.idle":"2022-10-02T10:59:27.325265Z","shell.execute_reply.started":"2022-10-02T10:59:10.094376Z","shell.execute_reply":"2022-10-02T10:59:27.324067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Curious - how many of the images have been presented as \"portrait\" and how many as \"landscape\" ?\n\nportraits  = len(train_csv[\"aspect_ratio\"][train_csv[\"aspect_ratio\"]<1])\nlandscapes = len(train_csv[\"aspect_ratio\"][train_csv[\"aspect_ratio\"]>1])\nsquares    = len(train_csv[\"aspect_ratio\"][train_csv[\"aspect_ratio\"]==1])\n\nportraits,landscapes, squares\n\n## No squares - that just doesn't sound right\n## Anyhow, I need to flip the landscapes over so whatever compression occurs will occur in one direction only.\n## Something tells me that is important.","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:59:27.327278Z","iopub.execute_input":"2022-10-02T10:59:27.327743Z","iopub.status.idle":"2022-10-02T10:59:27.339673Z","shell.execute_reply.started":"2022-10-02T10:59:27.327697Z","shell.execute_reply":"2022-10-02T10:59:27.338325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ACTION Let's add a flip boolean to the mix - we'll have to do it for test and train both.\n\ntest_csv[\"image_size\"]   = test_csv.image_path.apply(lambda x: Image.open(x).size)\ntest_csv[\"image_width\"]  = test_csv[\"image_size\"].apply(lambda x: int(x[0]))\ntest_csv[\"image_height\"] = test_csv[\"image_size\"].apply(lambda x: int(x[1]))\ntest_csv[\"aspect_ratio\"] = test_csv[\"image_width\"]/test_csv[\"image_height\"]\ndel test_csv[\"image_size\"]\n\ntrain_csv['flip'] = train_csv[\"aspect_ratio\"].apply(lambda x: 1 if x >1 else 0)\ntest_csv['flip']  = test_csv[\"aspect_ratio\"].apply(lambda x: 1 if x >1 else 0)\n\nprint(f\"Marking for flipping done for train. {train_csv['flip'].sum()} to-be-flippers identified.\")\nprint(f\"Marking for flipping done for test.  {test_csv['flip'].sum()} to-be-flippers identified.\")","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:59:27.341224Z","iopub.execute_input":"2022-10-02T10:59:27.341657Z","iopub.status.idle":"2022-10-02T10:59:27.436832Z","shell.execute_reply.started":"2022-10-02T10:59:27.341614Z","shell.execute_reply":"2022-10-02T10:59:27.435738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PREPROCESSING TIME - Let's cut those images to size\n\n## I'll be looking at this notebook from DARIEN SCHETTLER\n## https://www.kaggle.com/code/dschettler8845/mcsai-exploratory-data-analysis-baseline/notebook","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:59:27.43826Z","iopub.execute_input":"2022-10-02T10:59:27.439068Z","iopub.status.idle":"2022-10-02T10:59:27.443375Z","shell.execute_reply.started":"2022-10-02T10:59:27.439033Z","shell.execute_reply":"2022-10-02T10:59:27.442079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## But we get only one submission per day, so let me put one in, dummy though it may be - 50% probability either way\n\npatient_id = test_csv.patient_id\nCE  = [0.5]*len(patient_id) \nLAA = [0.5]*len(patient_id) \n  \nsubmission = pd.DataFrame(\n    {\n        \"patient_id\": patient_id,\n        \"CE\": CE,\n        \"LAA\": LAA,\n    }\n)","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:59:27.444862Z","iopub.execute_input":"2022-10-02T10:59:27.44524Z","iopub.status.idle":"2022-10-02T10:59:27.456333Z","shell.execute_reply.started":"2022-10-02T10:59:27.445197Z","shell.execute_reply":"2022-10-02T10:59:27.455185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Adjusting and writing to csv\n\nsubmission = submission.groupby(\"patient_id\").mean()\nsubmission = submission.round(6).reset_index()\ndisplay(submission)\nsubmission.to_csv(\"submission.csv\", index=False)\n\n# Why all these steps - I've put in the explanation in \n# https://www.kaggle.com/code/saurabhsawhney/mayo-beginners-how-to-make-a-submission","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:59:27.457694Z","iopub.execute_input":"2022-10-02T10:59:27.458071Z","iopub.status.idle":"2022-10-02T10:59:27.489045Z","shell.execute_reply.started":"2022-10-02T10:59:27.458028Z","shell.execute_reply":"2022-10-02T10:59:27.488025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To Do\n\n# Look at options for preparing the images for the modelling process","metadata":{"execution":{"iopub.status.busy":"2022-10-02T10:59:27.490784Z","iopub.execute_input":"2022-10-02T10:59:27.491165Z","iopub.status.idle":"2022-10-02T10:59:27.495462Z","shell.execute_reply.started":"2022-10-02T10:59:27.491132Z","shell.execute_reply":"2022-10-02T10:59:27.494291Z"},"trusted":true},"execution_count":null,"outputs":[]}]}