{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport sys\nimport io\nimport warnings\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '2'\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport tensorflow_io as tfio\nimport pickle\nimport subprocess \nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom tqdm.contrib.concurrent import thread_map\nfrom tqdm.asyncio import tqdm as async_tqdm\nfrom sklearn.model_selection import train_test_split\n\nprint(\"Tensorflow version \" + tf.__version__)\n\npd.set_option('display.max_colwidth', 200)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-28T05:50:45.748402Z","iopub.execute_input":"2023-03-28T05:50:45.748839Z","iopub.status.idle":"2023-03-28T05:50:52.46012Z","shell.execute_reply.started":"2023-03-28T05:50:45.748746Z","shell.execute_reply":"2023-03-28T05:50:52.459062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_SIZE = [768, 512]\nMAIN_PATH = '/kaggle/input/rsna-breast-cancer-detection/'\nTRAIN_PATH = MAIN_PATH + 'train_images/'\nTEST_PATH = MAIN_PATH + 'test_images/'\nSUBPR_PATH = '/kaggle/input/dicom2tfrecord/dicom2png_tfio.py'\nWORKING_PATH = '/kaggle/working/'\nARGS_FILENAME = WORKING_PATH + 'args.pkl'\nTESTING = False # True - 500 images into 5 tfrecords of 100 images; False - all images into tfrecords files of 500 images","metadata":{"execution":{"iopub.status.busy":"2023-03-28T05:50:52.461959Z","iopub.execute_input":"2023-03-28T05:50:52.462517Z","iopub.status.idle":"2023-03-28T05:50:52.469253Z","shell.execute_reply.started":"2023-03-28T05:50:52.462488Z","shell.execute_reply":"2023-03-28T05:50:52.468275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_df = pd.read_csv(MAIN_PATH + 'train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-03-28T05:50:52.47112Z","iopub.execute_input":"2023-03-28T05:50:52.472251Z","iopub.status.idle":"2023-03-28T05:50:52.582643Z","shell.execute_reply.started":"2023-03-28T05:50:52.472216Z","shell.execute_reply":"2023-03-28T05:50:52.581621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_df.info()\n# There are Nans in:\n# age         - fill with median\n\n# BIRADS      - 0 if the breast required follow-up, \n#             - 1 if the breast was rated as negative for cancer, and \n#             - 2 if the breast was rated as normal\n#             Only provided for train\n\n# density     - A rating for how dense the breast tissue is, with A being the least dense and D being the most dense. \n#             Extremely dense tissue can make diagnosis more difficult. Only provided for train","metadata":{"execution":{"iopub.status.busy":"2023-03-28T05:50:52.585013Z","iopub.execute_input":"2023-03-28T05:50:52.585865Z","iopub.status.idle":"2023-03-28T05:50:52.620238Z","shell.execute_reply.started":"2023-03-28T05:50:52.585827Z","shell.execute_reply":"2023-03-28T05:50:52.618848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# leave alone BIADRDS and density, cause them provided only for train \nfill_nans = {'age': meta_df.age.median()}\nprint(fill_nans)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T05:50:52.621603Z","iopub.execute_input":"2023-03-28T05:50:52.621871Z","iopub.status.idle":"2023-03-28T05:50:52.631841Z","shell.execute_reply.started":"2023-03-28T05:50:52.621848Z","shell.execute_reply":"2023-03-28T05:50:52.630805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dataset is unbalanced so use stratification\ntrain, validation = train_test_split(meta_df, test_size = 0.05, random_state=42, stratify=meta_df.cancer)\nprint('number of patients per dataset:')\nprint(' - train - ', len(train))\nprint(' - valid - ', len(validation))\n\nprint('average number of cancers per 100 patients:')\nprint(' - train - {:0.2f}'.format(train.cancer.mean()*100))\nprint(' - valid - {:0.2f}'.format(validation.cancer.mean()*100))","metadata":{"execution":{"iopub.status.busy":"2023-03-28T05:50:52.63319Z","iopub.execute_input":"2023-03-28T05:50:52.63363Z","iopub.status.idle":"2023-03-28T05:50:52.673947Z","shell.execute_reply.started":"2023-03-28T05:50:52.633599Z","shell.execute_reply":"2023-03-28T05:50:52.672695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This runs subprocesses for making tfrecords-files using tfio.decode_dicom_image\n# I didn't find another way to make tfio.decode_dicom_image shut up\n\ndef _parallel_subproc(input_data):\n    \"\"\"\n    Prepares file with data for subprocess, and starts subprocess which makes tfrecord-files \n    \n    Parameters:\n        input_data - tuple of two elements, 0 - serial number, 1 - chunk of dataframe\n    \"\"\"\n    i, chunk, path = input_data\n    # file name for current iteration\n    args_filename = path + f'args{i:05d}.pkl'\n    # write data for current iteration\n    with open(args_filename, 'wb') as file_args:\n        pickle.dump((chunk, IMAGE_SIZE, i, path), file_args, protocol=pickle.HIGHEST_PROTOCOL)\n    # run subprocess\n    pipe = subprocess.run([sys.executable, SUBPR_PATH],\n                              input=args_filename, stdout=subprocess.PIPE, stderr=subprocess.PIPE, encoding='utf-8')\n#     print('stdout from parent:\\n', pipe.stderr) #uncomment if something goes wrong\n#     print('stdout from parent:\\n', pipe.stderr) #uncomment for profiling","metadata":{"execution":{"iopub.status.busy":"2023-03-28T05:51:52.439392Z","iopub.execute_input":"2023-03-28T05:51:52.439752Z","iopub.status.idle":"2023-03-28T05:51:52.446537Z","shell.execute_reply.started":"2023-03-28T05:51:52.439722Z","shell.execute_reply":"2023-03-28T05:51:52.445722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nif TESTING: # only some files...\n    nrows_train = 500\n    nrows_valid = 50\n    chunksize = 25\nelse: # whole dataset...\n    nrows_train = len(train)\n    nrows_valid = len(validation)\n    chunksize = 500\n\ntotal_train = np.ceil(nrows_train / chunksize)\ntotal_valid = np.ceil(nrows_valid / chunksize)\n\n# creating the necessary directories\ntry:\n    os.mkdir('train')\nexcept FileExistsError: \n    pass\n\ntry:\n    os.mkdir('valid')\nexcept FileExistsError: \n    pass\n\nprint('prepearing train files...')\nwith open('meta_data_train.tmp', 'w') as tmp_file:\n    # fill nans and place our csv in memory for chunking\n    tmp_file = io.StringIO(train.fillna(fill_nans).to_csv(index=False))\n    # path to train files\n    path_train = WORKING_PATH + 'train/'\n    with pd.read_csv(tmp_file, nrows=nrows_train, chunksize=chunksize) as reader:\n        # use tqdm staff for parallelizing subprocesses \n        thread_map(_parallel_subproc, \n                   [(i, chunk, path_train) for i, chunk in enumerate(reader)],\n                   tqdm_class=async_tqdm, total=total_train, ncols=100)\n        \nprint('prepearing validation files...')\nwith open('meta_data_valid.tmp', 'w') as tmp_file:\n    # fill nans and place our csv in memory for chunking\n    tmp_file = io.StringIO(validation.fillna(fill_nans).to_csv(index=False))\n    # path to valid files\n    path_valid = WORKING_PATH + 'valid/'\n    with pd.read_csv(tmp_file, nrows=nrows_valid, chunksize=chunksize) as reader:\n        # use tqdm staff for parallelizing subprocesses \n        thread_map(_parallel_subproc, \n                   [(i, chunk, path_valid) for i, chunk in enumerate(reader)],\n                   tqdm_class=async_tqdm, total=total_valid, ncols=100)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T06:25:09.44803Z","iopub.execute_input":"2023-03-28T06:25:09.448476Z","iopub.status.idle":"2023-03-28T06:29:26.274681Z","shell.execute_reply.started":"2023-03-28T06:25:09.448441Z","shell.execute_reply":"2023-03-28T06:29:26.271882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm /kaggle/working/train/args*.pkl /kaggle/working/valid/args*.pkl /kaggle/working/meta_data*.tmp","metadata":{"execution":{"iopub.status.busy":"2023-03-28T06:30:43.227664Z","iopub.execute_input":"2023-03-28T06:30:43.228256Z","iopub.status.idle":"2023-03-28T06:30:43.518058Z","shell.execute_reply.started":"2023-03-28T06:30:43.228194Z","shell.execute_reply":"2023-03-28T06:30:43.515984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), # tensor encoded as bytestring\n        \"label\": tf.io.FixedLenFeature([], tf.int64),  # shape [] means single element\n        \"meta_data\": tf.io.RaggedFeature(tf.string),\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = tf.io.parse_tensor(example['image'], tf.string)\n    image = tf.image.decode_jpeg(image)\n    label = tf.cast(example['label'], tf.uint8)\n    meta_data = example['meta_data']\n    return image, label, meta_data\n\ndef load_dataset(filenames, labeled=True, ordered=False, augmentation=False):\n    # Read from TFRecords. For optimal performance, reading from multiple files at once and\n    # disregarding data order. Order does not matter since we will be shuffling the data anyway.\n\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False # disable order, increase speed\n\n    dataset = tf.data.TFRecordDataset(filenames) # automatically interleaves reads from multiple files\n    dataset = dataset.with_options(ignore_order) # uses data as soon as it streams in, rather than in its original order\n    dataset = dataset.map(read_unlabeled_tfrecord if not labeled \n                          else read_labeled_tfrecord_wa if augmentation else read_labeled_tfrecord)\n#     dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord)\n    # returns a dataset of (image, label) pairs if labeled=True or (image, id) pairs if labeled=False\n    return dataset\n\ndef get_training_dataset(augmentation=False):\n    dataset = load_dataset(tf.io.gfile.glob(path_train + '*.tfrecords'), \n                           labeled=True, augmentation=augmentation)\n#     dataset = dataset.repeat() # the training dataset must repeat for several epochs\n#     dataset = dataset.shuffle(2048)\n#     dataset = dataset.batch(BATCH_SIZE)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2023-03-28T06:31:40.934751Z","iopub.execute_input":"2023-03-28T06:31:40.935209Z","iopub.status.idle":"2023-03-28T06:31:40.949941Z","shell.execute_reply.started":"2023-03-28T06:31:40.935176Z","shell.execute_reply":"2023-03-28T06:31:40.94821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr_ds = get_training_dataset()","metadata":{"execution":{"iopub.status.busy":"2023-03-28T06:31:44.212134Z","iopub.execute_input":"2023-03-28T06:31:44.212546Z","iopub.status.idle":"2023-03-28T06:31:44.251155Z","shell.execute_reply.started":"2023-03-28T06:31:44.212517Z","shell.execute_reply":"2023-03-28T06:31:44.249967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 60))\nfor i, (image, label, md) in enumerate(tr_ds.take(50).as_numpy_iterator()):\n    meta_dict = pickle.loads(md[0])\n    ax = plt.subplot(10, 5, i + 1)\n    plt.imshow(image, cmap='gray')\n    #plt.title(int(label))\n    plt.title(f\"{meta_dict['patient_id']}/{meta_dict['image_id']}\")\n    plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2023-03-28T06:31:47.109114Z","iopub.execute_input":"2023-03-28T06:31:47.109504Z","iopub.status.idle":"2023-03-28T06:31:53.452857Z","shell.execute_reply.started":"2023-03-28T06:31:47.109473Z","shell.execute_reply":"2023-03-28T06:31:53.45113Z"},"trusted":true},"execution_count":null,"outputs":[]}]}