{"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.0.5"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In-progress. Need to convert entirety of notebooks into R for submissions to work.\n","metadata":{}},{"cell_type":"code","source":"# Still need to convert into R...\n## https://www.kaggle.com/code/aritrag/kerascv-starter-notebook-train\n## https://www.kaggle.com/code/aritrag/kerascv-starter-notebook-infer\n\n# Already converted\n##   https://www.kaggle.com/code/jakebrusca/rsna23-weighted-mean-baseline\n\n# \n##   https://www.linkedin.com/pulse/kaggle-rsna-2023-abdominal-trauma-detection-ivan-isaev-","metadata":{"execution":{"iopub.status.busy":"2023-09-27T23:59:41.99246Z","iopub.execute_input":"2023-09-27T23:59:41.993793Z","iopub.status.idle":"2023-09-27T23:59:42.003267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Needed update of R and requested: https://www.kaggle.com/discussions/product-feedback/443458\n\n# Added from the CRAN repository https://cran.r-project.org/bin/windows/contrib/4.4/espadon_1.4.1.zip\n# Installing an R package without Internet access after downloading the above, which was the latest \n#  version with a needed man file  : \"cannot open file 'man': No such file or directory\"\n# BUT failed: Kaggle R version too old ERROR: this R is version 4.0.5, package 'espadon' requires R >= 4.1.0\n# install_local('/kaggle/input/espadon/espadon', type = \"source\")\n# library(espadon)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Later consider: segmentation and classification\n## https://www.linkedin.com/pulse/kaggle-rsna-2023-abdominal-trauma-detection-ivan-isaev-","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# These packages were indicated by Bing AI to convert the standardize_pixel_array function in https://www.kaggle.com/code/aritrag/kerascv-starter-notebook-train\n# https://www.kaggle.com/competitions/data-science-bowl-2017/discussion/27649\nlibrary(bitops)\n# Also consider oro.dicom discussion at https://www.kaggle.com/competitions/data-science-bowl-2017/discussion/27649\nlibrary(oro.dicom)\nlibrary(imager)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"raw","source":"# Averaging pinnned KerasCV and Weighted Mean notebooks\n# Each has duplicate predictions for all 3 training set patients, so not likely to predict well ultimately\n\n# R notebook - package install\n\nlibrary(dplyr) \nlibrary(readr)\n## Seeking to reproduce sklearn.metrics.log_loss (Log loss, aka logistic loss or cross-entropy loss)\nlibrary(MLmetrics)\nlibrary(keras)","metadata":{"execution":{"iopub.status.busy":"2023-09-27T11:31:11.686936Z","iopub.execute_input":"2023-09-27T11:31:11.689706Z","iopub.status.idle":"2023-09-27T11:31:13.351141Z"}}},{"cell_type":"code","source":"## Keras package failed to load model and also failed efficient : reticulate to use Python code\n## The error message you’re seeing, ('Unrecognized keyword arguments:', dict_keys(['batch_shape'])), typically occurs when there’s a mismatch between the Keras version used to save the model and the one used to load it12345.\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KerasCV: Train","metadata":{}},{"cell_type":"code","source":"library(keras)\n\nConfig <- list(\n  SEED = 42,\n  IMAGE_SIZE = c(256, 256),\n  BATCH_SIZE = 64,\n  EPOCHS = 10,\n  TARGET_COLS = c(\n    \"bowel_injury\", \"extravasation_injury\",\n    \"kidney_healthy\", \"kidney_low\", \"kidney_high\",\n    \"liver_healthy\", \"liver_low\", \"liver_high\",\n    \"spleen_healthy\", \"spleen_low\", \"spleen_high\"\n  ),\n  AUTOTUNE = tensorflow::tf$data$AUTOTUNE\n)\n\nconfig <- Config\n\nset.seed(90)","metadata":{"execution":{"iopub.status.busy":"2023-10-02T23:56:41.87986Z","iopub.execute_input":"2023-10-02T23:56:41.881855Z","iopub.status.idle":"2023-10-02T23:56:41.908441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/rsna-atd-512x512-png-v2-dataset\"\n","metadata":{"execution":{"iopub.status.busy":"2023-10-03T00:12:31.171947Z","iopub.execute_input":"2023-10-03T00:12:31.173584Z","iopub.status.idle":"2023-10-03T00:12:31.199207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"library(keras)\n\nConfig <- list(\n  SEED = 42,\n  IMAGE_SIZE = c(256, 256),\n  BATCH_SIZE = 64,\n  EPOCHS = 10,\n  TARGET_COLS = c(\n    \"bowel_injury\", \"extravasation_injury\",\n    \"kidney_healthy\", \"kidney_low\", \"kidney_high\",\n    \"liver_healthy\", \"liver_low\", \"liver_high\",\n    \"spleen_healthy\", \"spleen_low\", \"spleen_high\"\n  ),\n  AUTOTUNE = tensorflow::tf$data$AUTOTUNE\n)\n\nconfig <- Config\n\nSEED = 42\n  IMAGE_SIZE = c(256, 256)\n  BATCH_SIZE = 64\n  EPOCHS = 10\n  TARGET_COLS = c(\n    \"bowel_injury\", \"extravasation_injury\",\n    \"kidney_healthy\", \"kidney_low\", \"kidney_high\",\n    \"liver_healthy\", \"liver_low\", \"liver_high\",\n    \"spleen_healthy\", \"spleen_low\", \"spleen_high\"\n  )\n  AUTOTUNE = tensorflow::tf$data$AUTOTUNE","metadata":{"execution":{"iopub.status.busy":"2023-10-03T00:14:49.94527Z","iopub.execute_input":"2023-10-03T00:14:49.947824Z","iopub.status.idle":"2023-10-03T00:14:49.984587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"library(dplyr)\nlibrary(readr)\nlibrary(tidyr)\nlibrary(caret)\n\n# Read the CSV file\ndataframe <- read_csv(paste0(BASE_PATH, \"/train.csv\"))\n\n# Create the image_path column\ndataframe <- dataframe %>%\n  mutate(image_path = paste0(BASE_PATH, \"/train_images/\",\n                             as.character(patient_id), \"/\",\n                             as.character(series_id), \"/\",\n                             as.character(instance_number), \".png\"))\n\n# Remove duplicates\ndataframe <- distinct(dataframe)\n\n# Create a grouping variable based on TARGET_COLS\ndataframe$group <- apply(dataframe[TARGET_COLS], 1, paste, collapse = \"_\")\n\n# Split the data into training and validation sets\nsplitIndex <- createDataPartition(dataframe$group, p = 0.8, list = FALSE)\ntrain_data <- dataframe[splitIndex, ]\nval_data <- dataframe[-splitIndex, ]\n\n# Remove the grouping variable\ntrain_data$group <- NULL\nval_data$group <- NULL\n","metadata":{"execution":{"iopub.status.busy":"2023-10-03T00:24:25.838399Z","iopub.execute_input":"2023-10-03T00:24:25.840888Z","iopub.status.idle":"2023-10-03T00:24:28.687035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nrow(dataframe)\nncol(dataframe)\n\nnrow(train_data) \nncol(train_data)\n\nnrow(val_data)\nncol(val_data)\n\n## 9612 21 and 2409 21 in match https://www.kaggle.com/code/rickpack/kerascv-starter-notebook-train\n## Close enough split between train and val","metadata":{"execution":{"iopub.status.busy":"2023-10-03T00:25:43.046158Z","iopub.execute_input":"2023-10-03T00:25:43.04769Z","iopub.status.idle":"2023-10-03T00:25:43.088832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"library(imager)\nlibrary(keras)\n\ndecode_image_and_label <- function(image_path, label) {\n  image <- load.image(image_path)\n  image <- resize(image, config$IMAGE_SIZE[1], config$IMAGE_SIZE[2])\n  image <- imager::as.data.frame(image) / 255\n  \n  label <- as.numeric(label)\n  labels <- list(label[1:2], label[3:6], label[7:10], label[11:14], label[15:18])\n  \n  list(image, labels)\n}\n\napply_augmentation <- function(images, labels) {\n  augmenter <- keras::layer_random_flip(mode = \"horizontal_and_vertical\") %>%\n    keras::layer_random_zoom(height_factor = 0.2, width_factor = 0.2)\n  \n  list(augmenter(images), labels)\n}\n\n## This sprang Error in `purrr::map2()`:\n## ! Can't recycle `.x` (size 9637) to match `.y` (size 106007).\n#build_dataset <- function(image_paths, labels) {\n#  ds <- purrr::map2(image_paths, labels, decode_image_and_label)\n  \n#  ds <- sample(ds, config$BATCH_SIZE * 10)\n  \n#  ds <- purrr::map(ds, apply_augmentation)\n  \n # ds\n#}\n\nbuild_dataset <- function(image_paths, labels) {\n  ds <- mapply(decode_image_and_label, image_paths, labels, SIMPLIFY = FALSE)\n  \n  ds <- sample(ds, config$BATCH_SIZE * 10)\n  \n  ds <- lapply(ds, apply_augmentation)\n  \n  ds\n}\n","metadata":{"execution":{"iopub.status.busy":"2023-10-03T01:04:21.790714Z","iopub.execute_input":"2023-10-03T01:04:21.796711Z","iopub.status.idle":"2023-10-03T01:04:21.837548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Assuming that train_data is a data frame and TARGET_COLS is a character vector\npaths <- as.character(train_data$image_path)\nlabels <- as.matrix(train_data[, TARGET_COLS])\n\n# Assuming build_dataset() is a function that takes image_paths and labels as arguments\nds <- build_dataset(image_paths = paths, labels = labels)","metadata":{"execution":{"iopub.status.busy":"2023-10-03T01:04:25.021811Z","iopub.execute_input":"2023-10-03T01:04:25.024394Z","iopub.status.idle":"2023-10-03T01:04:25.121192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class(ds)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-10-03T00:42:31.02161Z","iopub.execute_input":"2023-10-03T00:42:31.023369Z","iopub.status.idle":"2023-10-03T00:42:31.041565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming ds is a list of images and labels\nimages <- ds[[1]]\nlabels <- ds[[2]]\n\n# Get the shapes\nimage_shape <- dim(images)\nlabel_shapes <- lapply(labels, dim)\n\nlist(image_shape, label_shapes)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import necessary Python modules using import() function\nkeras <- import(\"keras\")\n\n# Load JSON model architecture\njson_file <- file(\"model_architecture.json\", \"r\")\nloaded_model_json <- readLines(json_file)\nclose(json_file)\n\n# Convert JSON model to Keras model\nloaded_model <- keras$models$model_from_json(loaded_model_json)\n\n# Load weights into new model\nloaded_model$load_weights(\"model_weights.h5\")\nprint(\"Loaded model from disk\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now you can use the model object in R\nmodel <- py$model","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"?load_model_weights_hdf5","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# BingAI conversion of Python https://www.kaggle.com/code/aritrag/kerascv-starter-notebook-train\n# Define a function to standardize the pixel array of a DICOM file\nstandardize_pixel_array <- function(dcm) {\n  # Correct DICOM pixel_array if PixelRepresentation == 1\n  pixel_array <- dcm$img\n  if (dcm$hdr$PixelRepresentation == 1) {\n    bit_shift <- dcm$hdr$BitsAllocated - dcm$hdr$BitsStored\n        dtype <- typeof(pixel_array) # Get the data type of the pixel array\n    new_array <- bitShiftL(pixel_array, bit_shift) # Left shift the pixel array by bit_shift\n    new_array <- as(new_array, dtype) # Convert the new array to the original data type\n    new_array <- bitShiftR(new_array, bit_shift) # Right shift the new array by bit_shift\n    pixel_array <- applyModalityLUT(new_array, dcm) # Apply the modality lookup table to the new array\n  }\n  return(pixel_array)\n}\n\n# Define a function to read an X-ray image from a DICOM file\nread_xray <- function(path, fix_monochrome = TRUE) {\n  dicom <- readDICOM(path) # Read the DICOM file\n  data <- standardize_pixel_array(dicom) # Standardize the pixel array\n  data <- data - min(data) # Subtract the minimum value from the data\n  data <- data / (max(data) + 1e-5) # Divide the data by the maximum value plus a small constant\n  if (fix_monochrome && dicom$hdr$PhotometricInterpretation == \"MONOCHROME1\") {\n    data <- 1.0 - data # Invert the data if it is monochrome\n  }\n  return(data)\n}\n\n# Define a function to resize and save an X-ray image from a DICOM file\nresize_and_save <- function(file_path) {\n  img <- read_xray(file_path) # Read the X-ray image from the DICOM file\n  h <- dim(img)[1] # Get the original height of the image\n  w <- dim(img)[2] # Get the original width of the image\n  img <- resize(img, config$RESIZE_DIM, config$RESIZE_DIM, method = \"bilinear\") # Resize the image using bilinear interpolation\n  img <- as.cimg(img) # Convert the image to a cimg object\n  img <- img * 255 # Multiply the image by 255\n  img <- as.integer(img) # Convert the image to an integer vector\n  \n  sub_path <- strsplit(file_path, \"/\", fixed = TRUE)[[1]][-c(1:4)] # Get the sub path of the file path\n  sub_path[length(sub_path)] <- gsub(\".dcm\", \".png\", sub_path[length(sub_path)]) # Replace .dcm with .png in the file name\n  infos <- sub_path # Get the information from the sub path\n  pid <- infos[length(infos) - 2] # Get the patient ID\n  sid <- infos[length(infos) - 1] # Get the study ID\n  iid <- infos[length(infos)] # Get the image ID\n  iid <- gsub(\".png\", \"\", iid) # Remove .png from the image ID\n  new_path <- file.path(IMAGE_DIR, sub_path) # Create a new path for saving the image\n  dir.create(dirname(new_path), recursive = TRUE, showWarnings = FALSE) # Create directories if they do not exist\n  imwrite(img, new_path) # Write the image to the new path\n}\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"# Only requires the training target data. \ny_train <- read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List of Targets\nInjuries = c('bowel_healthy', 'bowel_injury', \n            'extravasation_healthy', 'extravasation_injury', \n            'kidney_healthy', 'kidney_low', 'kidney_high', \n            'liver_healthy', 'liver_low', 'liver_high', \n            'spleen_healthy', 'spleen_low', 'spleen_high', \n            'any_injury')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Target EDA","metadata":{}},{"cell_type":"code","source":"summary(y_train[,Injuries])\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Score","metadata":{}},{"cell_type":"code","source":"## need to normalize? \n## Need to get logicDT package - https://search.r-project.org/CRAN/refmans/logicDT/html/calcNCE.html\n##  defaults to normalizing : https://scikit-learn.org/stable/modules/generated/sklearn.metrics.log_loss.html#:~:text=Log%20loss%2C%20aka%20logistic%20loss,for%20its%20training%20data%20y_true%20.\n## If true, return the mean loss per sample. Otherwise, return the sum of the per-sample losses.\nnormalize_probabilities_to_one <- function(df, group_columns) {\n    row_totals = apply(y_train[,Injuries], 2, sum)\n    if(min(row_totals) == 0) {\n        stop(\"All rows must contain at least one non-zero prediction\")        \n    }\n    y_train[,Injuries] <- y_train[,Injuries] / row_totals\n    invisible(df)\n}\n\nscore <- function(solution, submission, row_id_column_name) {\n    # need to add : # Run basic QC checks on the inputs\n    \n    # Calculate the label group log losses\n    binary_targets = c('bowel', 'extravasation')\n    triple_level_targets = c('kidney', 'liver', 'spleen')\n    all_target_categories = c(binary_targets, triple_level_targets)\n    \n    label_group_losses <- as.numeric()\n    \n    for (category in all_target_categories) {\n        if (category %in% binary_targets) {\n            col_group = c(paste0(category, \"_healthy\"), paste0(category, \"_injury\"))\n         } else {\n            col_group = c(paste0(category, \"_healthy\"), paste0(category, \"_low\"), paste0(category, \"_high\"))            \n        }\n    }\n    \n    solution <- normalize_probabilities_to_one(solution, col_group)\n    \n    for (col in col_group) {\n            if (!(col %in% colnames(submission))) {\n                stop(paste(\"Missing submission column :\", col))\n            }\n        }\n    submission <- normalize_probabilities_to_one(submission, col_group)\n\n    sample_weight <- solution %>% select(all_of(paste0(category, \"_weight\"))) %>% pull()\n        \n    # mlmetrics version\n    log_loss <- LogLoss(y_pred = as.numeric(unlist(sample_weight * (submission %>% select(all_of(col_group))))),\n                        y_true = as.numeric(unlist(sample_weight * (solution %>% select(all_of(col_group)))))\n                       )  \n    \n    print(sample_weight)\n    print(submission %>% select(all_of(col_group)))\n    \n    # base R version found on Stack Overflow - is not working => yields NaN\n    # logLoss2 <- function(pred, actual){\n    #              -mean(actual * log(pred) + (1 - actual) * log(1 - pred))\n     #           }\n    #log_loss <- logLoss2(pred = as.numeric(unlist(sample_weight * (submission %>% select(all_of(col_group))))),\n    #                    actual = as.numeric(unlist(sample_weight * (solution %>% select(all_of(col_group)))))\n    #                   )            \n    # print(paste('log_loss =', log_loss))\n    \n    label_group_losses <- c(label_group_losses, log_loss)\n    \n    sample_weight_any_injury <- solution %>% select(any_injury_weight) %>% pull()\n                \n    # Derive a new any_injury label by taking the max of 1 - p(healthy) for each label group\n    healthy_cols = paste0(all_target_categories, '_healthy')\n    any_injury_labels = 1 - apply(solution[, healthy_cols], 2, max)\n    any_injury_predictions = 1 - apply(submission[, healthy_cols], 2, max)\n    any_injury_loss = LogLoss(y_pred = as.numeric(unlist(sample_weight * (submission %>% select(all_of(col_group))))),\n                              y_true = as.numeric(unlist(sample_weight * (solution %>% select(all_of(col_group)))))\n                             )                    \n\n    label_group_losses <- c(label_group_losses, any_injury_loss)\n    return (mean(label_group_losses))\n}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign the appropriate weights to each category\ncreate_training_solution <- function(y_train) {\n  sol_train = y_train    \n  \n  # bowel healthy|injury sample weight = 1|2\n  sol_train <- sol_train %>%\n    mutate(bowel_weight = case_when(\n      bowel_injury == 1 ~ 2,\n      TRUE ~ 1)\n    )\n      \n  # extravasation healthy/injury sample weight = 1|6\n  sol_train <- sol_train %>%\n    mutate(extravasation_weight = case_when(\n      extravasation_injury == 1 ~ 6,\n      TRUE ~ 1)\n    )\n      \n  # kidney healthy|low|high sample weight = 1|2|4\n  sol_train <- sol_train %>%\n    mutate(kidney_weight = case_when(\n      kidney_low == 1 ~ 2,\n      kidney_high == 1 ~ 4,\n      TRUE ~ 1)\n    )\n      \n  # liver healthy|low|high sample weight = 1|2|4\n  sol_train <- sol_train %>%\n    mutate(liver_weight = case_when(\n      liver_low == 1 ~ 2,\n      liver_high == 1 ~ 4,\n      TRUE ~ 1)\n    )\n     \n  # spleen healthy|low|high sample weight = 1|2|4\n  sol_train <- sol_train %>%\n    mutate(spleen_weight = case_when(\n      spleen_low == 1 ~ 2,\n      spleen_high == 1 ~ 4,\n      TRUE ~ 1)\n    )\n  \n  # any healthy|injury sample weight = 1|6\n  sol_train <- sol_train %>% \n    mutate(any_injury_weight = case_when(\n      any_injury == 1 ~ 6,\n      TRUE ~ 1))\n}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"solution_train = create_training_solution(y_train)\n\n# predict a constant using the mean of the training data\ny_pred = y_train\ncols_train <- colnames(y_train)\n\ny_train_means <- apply(y_train %>% select(all_of(Injuries)), 2, mean)\nfor (col_use in cols_train) {\n    new_value = y_train_means[col_use]\n    y_pred <- y_pred %>%\n      mutate(!!col_use := new_value)\n}\n\n# y_pred[,Injuries] = \nno_scale_score = score(solution_train,y_pred,'patient_id')\n# This is 2x the 0.78 in https://www.kaggle.com/code/rickpack/rsna23-weighted-mean-baseline\nprint(paste('Training score without scaling (copied notebook = 0.78):', no_scale_score))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Group by different sample weights\nscale_by_2 = c('bowel_injury','kidney_low','liver_low','spleen_low')\nscale_by_4 = c('kidney_high','liver_high','spleen_high')\nscale_by_6 = c('extravasation_injury','any_injury')\nscale_healthy = c('bowel_healthy', 'extravasation_healthy', 'kidney_healthy', 'liver_healthy', 'spleen_healthy')\n# Scale factors based on described metric \nsf_2 = 2\nsf_4 = 4\nsf_6 = 6\n\n# The score function deletes the ID column so we remake it\n# solution_train = create_training_solution(y_train)\n\n## Don't think this needed\n# Reset the prediction\n#y_pred = y_train.copy()\n#y_pred[Injuries] = y_train[Injuries].mean().tolist()\n\n# Scale each target \ny_pred2 <- y_pred %>% mutate(across((!!scale_by_2), ~.x * 2))\ny_pred2 <- y_pred2 %>% mutate(across((!!scale_by_4), ~.x * 4))\ny_pred2 <- y_pred2 %>% mutate(across((!!scale_by_6), ~.x * 6))\n\n# y_pred[scale_healthy] *=scale_h\n\nweight_scale_score = score(solution_train,y_pred2,'patient_id')\nprint(paste('Training score with weight scaling: ', weight_scale_score))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load submission datasets\n\n# Submissions using my version of https://www.kaggle.com/code/aritrag/kerascv-starter-notebook-infer\n#   My score was a horrible 12.16, author ARITRA ROY GOSTHIPATY reported 7.4 (random variation, didn't we use the same seed?)\nsub_keras_starter = read_csv('/kaggle/input/kerascv-starter-notebook-infer/submission.csv')\nsub_keras_starter = sub_keras_starter %>%\n  arrange(patient_id)\n# Submissions using the weighted means of https://www.kaggle.com/code/rickpack/rsna23-weighted-mean-baseline-scale-adj-at-end\nsub_weighted_mean = read_csv('/kaggle/input/rsna23-weighted-mean-baseline-scale-adj-at-end/submission.csv')\nsub_weighted_mean = sub_weighted_mean %>%\n  arrange(patient_id)\nsub_weighted_mean = sub_weighted_mean %>% \n  select(-any_injury)\n# arrange sorted so both submission files have the same patient_id sequence\n#  capture patient_id in a column\npatient_id_col = sub_weighted_mean$patient_id","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"str(sub_keras_starter)\nstr(sub_weighted_mean)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop patient_id (so all rows, comma, second column and to the right)\nsub_keras_starter_nopatient_id <- sub_keras_starter[,2:ncol(sub_keras_starter)]\nsub_weighted_mean_nopatient_id <- sub_weighted_mean[,2:ncol(sub_weighted_mean)]\n\n# Then take mean of each column - this is easy using R\naverage_of_subs = (sub_keras_starter_nopatient_id + 19 * sub_weighted_mean_nopatient_id) / 20\n\n# Now add back patient_id with columnbind function cbind()\nsub_average_of_subs = cbind(patient_id_col, average_of_subs)\nsub_average_of_subs = sub_average_of_subs %>%\n  rename(patient_id = patient_id_col)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"together <- left_join(sub_keras_starter, sub_weighted_mean , by = \"patient_id\")\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the sample submission CSV file\nsample_submission = read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/sample_submission.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_avg <- t(data.frame(\n    apply(y_pred2 %>% select(-patient_id), 2, mean)))\nrownames(y_pred_avg) <- NA\n\nsubmission <- bind_cols(\n    sample_submission %>% select(patient_id),\n    y_pred_avg)\n\nwrite_csv(submission, 'submission.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}