{"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.0.5"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Let's explore the structured part of the training data.\n### Short overview: We have 754 rows (corresponding to 754 training images) in the training data covering 632 patients across 11 medical centers. For each patient we have between 1 and 5 images available.","metadata":{}},{"cell_type":"code","source":"# packages\nlibrary(tidyverse)\nlibrary(RColorBrewer)","metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","execution":{"iopub.status.busy":"2022-07-10T14:31:02.470588Z","iopub.execute_input":"2022-07-10T14:31:02.478367Z","iopub.status.idle":"2022-07-10T14:31:02.49445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show files\nlist.files(path = '../input/mayo-clinic-strip-ai')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:02.498299Z","iopub.execute_input":"2022-07-10T14:31:02.499977Z","iopub.status.idle":"2022-07-10T14:31:02.520888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# configure default plot size\nplot_w <- 8\nplot_h <- 5\noptions(repr.plot.width = plot_w, repr.plot.height = plot_h)\n\n# configure data frame display\noptions(repr.matrix.max.cols=500)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:02.524445Z","iopub.execute_input":"2022-07-10T14:31:02.526022Z","iopub.status.idle":"2022-07-10T14:31:02.547002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Set EDA","metadata":{}},{"cell_type":"code","source":"# load training data\ndf <- read.csv('../input/mayo-clinic-strip-ai/train.csv')\n\n# some conversions\ndf$image_id <- as.factor(df$image_id)\ndf$center_id <- as.factor(df$center_id)\ndf$patient_id <- as.factor(df$patient_id)\ndf$label <- as.factor(df$label)\n\n# number of observations (~ images)\nn_train <- nrow(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:02.551198Z","iopub.execute_input":"2022-07-10T14:31:02.552864Z","iopub.status.idle":"2022-07-10T14:31:02.583082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat('Number of images [train] :', n_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:02.587099Z","iopub.execute_input":"2022-07-10T14:31:02.588853Z","iopub.status.idle":"2022-07-10T14:31:02.605849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# overview\nstr(df, list.len=1000)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:02.609533Z","iopub.execute_input":"2022-07-10T14:31:02.611973Z","iopub.status.idle":"2022-07-10T14:31:02.634848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Centers","metadata":{}},{"cell_type":"code","source":"# center distribution (counting images, not patients!)\nsummary(df$center_id)\nplot(df$center_id, main='center_id'); grid()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:02.638557Z","iopub.execute_input":"2022-07-10T14:31:02.640842Z","iopub.status.idle":"2022-07-10T14:31:02.721691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show percentages also\nround(summary(df$center_id) / n_train,4)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:02.725873Z","iopub.execute_input":"2022-07-10T14:31:02.728162Z","iopub.status.idle":"2022-07-10T14:31:02.747956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### => More than one third of observations stems from center with id=11 alone!","metadata":{}},{"cell_type":"markdown","source":"### Patients by Center","metadata":{}},{"cell_type":"code","source":"# to avoid duplicates keep only one row per patient (image_num=0) \ndf_1 <- df[df$image_num==0,]\nn_train_1 <- nrow(df_1)\n\n# center distribution, patients occur only once\nsummary(df_1$center_id)\nplot(df_1$center_id, main='center_id w/o duplicates of patients'); grid()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:02.751606Z","iopub.execute_input":"2022-07-10T14:31:02.753068Z","iopub.status.idle":"2022-07-10T14:31:02.836937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat('Number of patients [train] :', n_train_1)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:02.839402Z","iopub.execute_input":"2022-07-10T14:31:02.841069Z","iopub.status.idle":"2022-07-10T14:31:02.859869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show percentages also\nround(summary(df_1$center_id) / n_train_1,4)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:02.862154Z","iopub.execute_input":"2022-07-10T14:31:02.863598Z","iopub.status.idle":"2022-07-10T14:31:02.880005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Patients and Images","metadata":{}},{"cell_type":"code","source":"# patient distribution\noptions(repr.plot.width = 18, repr.plot.height = plot_h)\nplot(df$patient_id, main='patient_id'); grid()\noptions(repr.plot.width = plot_w, repr.plot.height = plot_h)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:02.883418Z","iopub.execute_input":"2022-07-10T14:31:02.884876Z","iopub.status.idle":"2022-07-10T14:31:03.032925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### => Patients can have up to 5 observations!","metadata":{}},{"cell_type":"code","source":"# frequency table for patients\npat_tab <- as.data.frame(table(df$patient_id))\ncolnames(pat_tab) <- c('patient_id','Freq')\n# sort by frequency\npat_tab <- pat_tab[order(pat_tab$Freq, decreasing = TRUE),]\n# show top 10 entries\nhead(pat_tab,10)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.037037Z","iopub.execute_input":"2022-07-10T14:31:03.039045Z","iopub.status.idle":"2022-07-10T14:31:03.073606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# extract the 4 patients with frequency = 5\npat_tab_5 <- pat_tab[pat_tab$Freq==5,]\npat_tab_5_ids <- pat_tab_5$patient_id\n# show corresponding records\ndf[df$patient_id %in% pat_tab_5_ids,]","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.077406Z","iopub.execute_input":"2022-07-10T14:31:03.078976Z","iopub.status.idle":"2022-07-10T14:31:03.121454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for the sake of completeness show all patients (id) with multiple images\npat_tab_mult <- pat_tab[pat_tab$Freq>1,]\npat_tab_mult_ids <- as.character(pat_tab_mult$patient_id)\nprint(pat_tab_mult_ids)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.125339Z","iopub.execute_input":"2022-07-10T14:31:03.127661Z","iopub.status.idle":"2022-07-10T14:31:03.152204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image numbers (counting starts at 0!)\nsummary(as.factor(df$image_num))\nplot(as.factor(df$image_num), main='image_num')\ngrid()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.155361Z","iopub.execute_input":"2022-07-10T14:31:03.156981Z","iopub.status.idle":"2022-07-10T14:31:03.244195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### From the numbers above we can deduce (looking at the increments) that we have 4 patients with 5 images, 4 patients with 4 images, 13 patients with 3 images, 68 patients with 2 images and 543 patients with 1 image.","metadata":{}},{"cell_type":"markdown","source":"### Label Distributions","metadata":{}},{"cell_type":"code","source":"# label distribution\nsummary(df$label)\nplot(df$label, main='label')\ngrid()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.24779Z","iopub.execute_input":"2022-07-10T14:31:03.249316Z","iopub.status.idle":"2022-07-10T14:31:03.323751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Let's again remove multiple images to have an unbiased view. Labels do not change across different images for a given patient:","metadata":{}},{"cell_type":"code","source":"# label distribution - patients instead of images\nsummary(df_1$label)\nplot(df$label, main='label (one per patient)')\ngrid()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.331611Z","iopub.execute_input":"2022-07-10T14:31:03.333004Z","iopub.status.idle":"2022-07-10T14:31:03.406621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show label distribution by center\ntab_dc <- table(df_1$center_id, df_1$label)\ntab_dc","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.409024Z","iopub.execute_input":"2022-07-10T14:31:03.410439Z","iopub.status.idle":"2022-07-10T14:31:03.427782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# normalize each row\ntab_dc_perc <- tab_dc / rowSums(tab_dc)\ntab_dc_perc","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.430168Z","iopub.execute_input":"2022-07-10T14:31:03.43159Z","iopub.status.idle":"2022-07-10T14:31:03.451504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualize label distributions by center\noptions(repr.plot.width = 10, repr.plot.height = 8)\ncorrplot::corrplot(tab_dc_perc, is.corr = F, \n                   col = corrplot::COL2('RdBu', 16),\n                   cl.pos='n')\noptions(repr.plot.width = plot_w, repr.plot.height = plot_h)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.455125Z","iopub.execute_input":"2022-07-10T14:31:03.456936Z","iopub.status.idle":"2022-07-10T14:31:03.57055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### => Especially center 3 shows a quite different label distribution.","metadata":{}},{"cell_type":"markdown","source":"# Let's have a look on the test set and the sample submission as well","metadata":{}},{"cell_type":"code","source":"# sample submission\ndf_sub_sample <- read.csv('../input/mayo-clinic-strip-ai/sample_submission.csv')\ndf_sub_sample","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.574784Z","iopub.execute_input":"2022-07-10T14:31:03.576617Z","iopub.status.idle":"2022-07-10T14:31:03.60286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load test set\ndf_test <- read.csv('../input/mayo-clinic-strip-ai/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.6054Z","iopub.execute_input":"2022-07-10T14:31:03.606839Z","iopub.status.idle":"2022-07-10T14:31:03.624137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.628187Z","iopub.execute_input":"2022-07-10T14:31:03.629995Z","iopub.status.idle":"2022-07-10T14:31:03.651879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for testing purpose only: replace dummy test set with training data\n# df_test <- df\n# df_test$center_id <- as.integer(as.character(df_test$center_id))\n# df_test$label <- NULL\n# df_test$center_id[3] <- NA # add artificial missing\n# df_test$center_id[6] <- 15 # add artificial unseen level\n# df_test","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.654359Z","iopub.execute_input":"2022-07-10T14:31:03.65581Z","iopub.status.idle":"2022-07-10T14:31:03.667042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# add columns for predictions to test set\ndf_test$CE <- 0\ndf_test$LAA <- 0","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.669629Z","iopub.execute_input":"2022-07-10T14:31:03.671041Z","iopub.status.idle":"2022-07-10T14:31:03.687082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# number of rows in test set\nn_test <- nrow(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.690927Z","iopub.execute_input":"2022-07-10T14:31:03.692587Z","iopub.status.idle":"2022-07-10T14:31:03.705121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calc predictions\nfor (i in 1:n_test) {\n    # get center (force conversion to numeric)\n    current_center <- as.numeric(as.character(df_test$center_id[i]))\n    if (is.na(current_center)) current_center <- 0 # fix potential NAs\n    # if center id in observed range lookup percentage from table above\n    if ((current_center >=1) & (current_center <= 11)) {\n        p_CE <- tab_dc_perc[current_center,'CE']\n        p_LAA <- tab_dc_perc[current_center,'LAA']        \n    } else {\n        # fallback scenario in case center_id is unknown\n        p_CE <- 0.5\n        p_LAA <- 0.5\n    }\n    # write values into df_test table\n    df_test$CE[i] <- p_CE\n    df_test$LAA[i] <- p_LAA\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.71333Z","iopub.execute_input":"2022-07-10T14:31:03.714819Z","iopub.status.idle":"2022-07-10T14:31:03.743368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show results\ndf_test","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.7472Z","iopub.execute_input":"2022-07-10T14:31:03.748952Z","iopub.status.idle":"2022-07-10T14:31:03.773785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary(df_test[,c('CE','LAA')])","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.776206Z","iopub.execute_input":"2022-07-10T14:31:03.777626Z","iopub.status.idle":"2022-07-10T14:31:03.796005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# carve out submission file from the extended test set\ndf_sub <- df_test[,c('patient_id','CE','LAA')]\n\n# remove duplicates (patients can have more than 1 image)!!!\ndf_sub <- df_sub[!duplicated(df_sub),]","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.79972Z","iopub.execute_input":"2022-07-10T14:31:03.80147Z","iopub.status.idle":"2022-07-10T14:31:03.817095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# and save submission\nwrite_delim(df_sub, file='submission.csv', delim=',')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.819891Z","iopub.execute_input":"2022-07-10T14:31:03.8214Z","iopub.status.idle":"2022-07-10T14:31:03.839543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# infrastructure details\nsessionInfo()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T14:31:03.843352Z","iopub.execute_input":"2022-07-10T14:31:03.845146Z","iopub.status.idle":"2022-07-10T14:31:03.890881Z"},"trusted":true},"execution_count":null,"outputs":[]}]}