{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook, `.dcm` images are converted to `.png` compressed into a tfrecord including data about patient information like `age`, `site_id`, `cancer` etc...\n\nThis notebook is based on the **RSNA Breast Cancer Detection** competition. The data for the competition includes over 50000 medical images of breast cancer and a `.csv` file that contains other useful information about the patient.\n","metadata":{}},{"cell_type":"code","source":"#install libraries required for pydicom (pixel_array)\n\n! pip install python-gdcm\n! pip install pylibjpeg","metadata":{"execution":{"iopub.status.busy":"2023-01-08T21:31:34.160262Z","iopub.execute_input":"2023-01-08T21:31:34.160789Z","iopub.status.idle":"2023-01-08T21:32:00.171883Z","shell.execute_reply.started":"2023-01-08T21:31:34.16075Z","shell.execute_reply":"2023-01-08T21:32:00.170719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The `python-gdcm` and `pylibjpeg` libraries are required for reading the image data from `.dcm` files. Although some of the images which can be read without these libraries, most through an exception that these libraries are required. ","metadata":{}},{"cell_type":"code","source":"#import libraries\nimport pandas as pd\nimport pydicom as dicom #to read .dcm images \nimport matplotlib.pyplot as plt\nimport cv2\n\nimport tensorflow as tf\nimport tensorflow_io as tfio\nimport numpy as np\n\nimport os\n\nimport pylibjpeg\n\n%env TF_CPP_MIN_LOG_LEVEL=2\n\nimport warnings\n\n# Ignore warning messages that match a given pattern\nwarnings.filterwarnings(\"ignore\", \"Cleanup called...\")","metadata":{"execution":{"iopub.status.busy":"2023-01-08T21:32:00.174289Z","iopub.execute_input":"2023-01-08T21:32:00.175208Z","iopub.status.idle":"2023-01-08T21:32:08.08205Z","shell.execute_reply.started":"2023-01-08T21:32:00.175165Z","shell.execute_reply":"2023-01-08T21:32:08.080882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-08T21:32:08.084203Z","iopub.execute_input":"2023-01-08T21:32:08.084822Z","iopub.status.idle":"2023-01-08T21:32:08.249601Z","shell.execute_reply.started":"2023-01-08T21:32:08.084787Z","shell.execute_reply":"2023-01-08T21:32:08.248356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_dir(patient_id, image_id):\n    return f'/kaggle/input/rsna-breast-cancer-detection/train_images/{patient_id}/{image_id}.dcm'","metadata":{"execution":{"iopub.status.busy":"2023-01-08T21:32:08.252406Z","iopub.execute_input":"2023-01-08T21:32:08.25277Z","iopub.status.idle":"2023-01-08T21:32:08.258187Z","shell.execute_reply.started":"2023-01-08T21:32:08.252739Z","shell.execute_reply":"2023-01-08T21:32:08.256624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['image_path'] = train[['patient_id', 'image_id']].apply(lambda x: get_dir(x['patient_id'], x['image_id']), axis=1)\ntrain['image_path'][0]","metadata":{"execution":{"iopub.status.busy":"2023-01-08T21:32:08.260291Z","iopub.execute_input":"2023-01-08T21:32:08.260842Z","iopub.status.idle":"2023-01-08T21:32:09.142161Z","shell.execute_reply.started":"2023-01-08T21:32:08.260753Z","shell.execute_reply":"2023-01-08T21:32:09.140883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"According to the competition page, some of the features in the `train.csv` file are not present in `test.csv`, so they are removed there. I also remove `image_path` as I only want features that would be in the tfrecord file.","metadata":{}},{"cell_type":"code","source":"#drop columns only provided for train dataset\ny = train['cancer']\ntrain_df = train.drop(columns=['biopsy', 'invasive', 'BIRADS', 'density', 'difficult_negative_case', 'image_path'])\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-08T21:32:09.14436Z","iopub.execute_input":"2023-01-08T21:32:09.144818Z","iopub.status.idle":"2023-01-08T21:32:09.17092Z","shell.execute_reply.started":"2023-01-08T21:32:09.144771Z","shell.execute_reply":"2023-01-08T21:32:09.169665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fill null values with -1\ntrain_df.fillna(-1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T21:32:09.172834Z","iopub.execute_input":"2023-01-08T21:32:09.173334Z","iopub.status.idle":"2023-01-08T21:32:09.187316Z","shell.execute_reply.started":"2023-01-08T21:32:09.173271Z","shell.execute_reply":"2023-01-08T21:32:09.186371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['age'] = train_df['age'].astype('int64') #convert age from float64 to int64\ntrain_df = pd.get_dummies(train_df, columns=['machine_id', 'view']) #one-hot encode 'machine_id' and 'view' \ntrain_df['laterality'] = train_df['laterality'].replace({'L':0, 'R':1}) #change 'laterality' to numbers 0 and 1\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-08T21:32:09.188968Z","iopub.execute_input":"2023-01-08T21:32:09.189335Z","iopub.status.idle":"2023-01-08T21:32:09.257069Z","shell.execute_reply.started":"2023-01-08T21:32:09.189278Z","shell.execute_reply":"2023-01-08T21:32:09.255851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfrecords_dir = \"tfrecords\"\n\nNUM_SAMPLES = 10000\nnum_tfrecords = max(train.shape) // NUM_SAMPLES\nif max(train.shape) % NUM_SAMPLES:\n    num_tfrecords += 1  # add one record if there are any remaining samples\n\nif not os.path.exists(tfrecords_dir):\n    os.makedirs(tfrecords_dir)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T21:32:49.425392Z","iopub.execute_input":"2023-01-08T21:32:49.425835Z","iopub.status.idle":"2023-01-08T21:32:49.432511Z","shell.execute_reply.started":"2023-01-08T21:32:49.425795Z","shell.execute_reply":"2023-01-08T21:32:49.431152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create Helper functions for TFrecord\ndef image_feature(value):\n    \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n    return tf.train.Feature(\n        bytes_list=tf.train.BytesList(value=[tf.io.encode_png(value).numpy()])\n    )\n\ndef int64_feature(value):\n    \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n    return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))\n\ndef create_example(image, row):\n    \"\"\"Write examples\"\"\"\n    feature = {\n        \"image\": image_feature(image),\n    }\n    for column in row.keys():\n        feature[f'{column}']=int64_feature(row[f'{column}'].values)        \n    return tf.train.Example(features=tf.train.Features(feature=feature))","metadata":{"execution":{"iopub.status.busy":"2023-01-08T21:32:53.509467Z","iopub.execute_input":"2023-01-08T21:32:53.509845Z","iopub.status.idle":"2023-01-08T21:32:53.518428Z","shell.execute_reply.started":"2023-01-08T21:32:53.509813Z","shell.execute_reply":"2023-01-08T21:32:53.517059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import concurrent.futures\nimport time\nimport tqdm\n\ndef process_image(image_path):\n    \"\"\"Read Image from path and resize it to (224, 224)\"\"\"\n    image_id = int(image_path.split('/')[-1].split('.')[0])\n    ds = dicom.dcmread(image_path)\n    image = cv2.resize(ds.pixel_array, [224, 224]).reshape(224, 224, 1)\n    row = train_df.loc[train_df['image_id'] == image_id]\n    example = create_example(image, row)\n    return example.SerializeToString()\n\ndef create_tfrecord(tfrec_num, samples, num_samples):\n    \"\"\"Write images and other patient data into tfrecord\"\"\"\n    tf_dir = ''\n    if num_samples == NUM_SAMPLES:\n        tf_dir = tfrecords_dir + \"/file_%.2i-%i.tfrec\" % ((tfrec_num * num_samples), ((tfrec_num + 1) * num_samples))\n    else:\n        tf_dir = tfrecords_dir + \"/file_%.2i-%i.tfrec\" % ((tfrec_num * num_samples), (tfrec_num * num_samples) + num_samples)\n    with tf.io.TFRecordWriter(\n        tf_dir\n    ) as writer:\n        for serialized_example in tqdm.tqdm(samples):\n            writer.write(serialized_example)\n\nnum_cores = 4\nimage_paths = train['image_path']\n\nprint('Start Processing Images into TFrecord')\n\nwith concurrent.futures.ProcessPoolExecutor(max_workers=num_cores) as executor:\n    for tfrec_num in range(num_tfrecords):\n        start_time = time.time()\n        samples = image_paths[(tfrec_num * NUM_SAMPLES) : ((tfrec_num + 1) * NUM_SAMPLES)]\n        serialized_examples = executor.map(process_image, samples)\n        create_tfrecord(tfrec_num, serialized_examples, len(samples))\n        print(f'time taken was {time.time() - start_time} seconds')\n","metadata":{"execution":{"iopub.status.busy":"2023-01-08T21:33:38.236123Z","iopub.execute_input":"2023-01-08T21:33:38.236543Z","iopub.status.idle":"2023-01-08T21:34:03.954439Z","shell.execute_reply.started":"2023-01-08T21:33:38.23651Z","shell.execute_reply":"2023-01-08T21:34:03.953051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def parse_tfrecord_fn(example):\n    \"\"\"Function to read examples from TFrecord\"\"\"\n    feature_description = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n    }\n    for column in row.keys():\n        if column not in ['patient_id', 'image_id']:\n            feature_description[f'{column}']=tf.io.FixedLenFeature([], tf.int64)\n    example = tf.io.parse_single_example(example, feature_description)\n    example[\"image\"] = tf.io.decode_png(example[\"image\"])\n    return example","metadata":{"execution":{"iopub.status.busy":"2023-01-08T20:55:00.354947Z","iopub.execute_input":"2023-01-08T20:55:00.355341Z","iopub.status.idle":"2023-01-08T20:55:00.363027Z","shell.execute_reply.started":"2023-01-08T20:55:00.355291Z","shell.execute_reply":"2023-01-08T20:55:00.361639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Extract the clean `train_df` dataframe \ntrain_df.to_csv('train_df.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T21:36:08.530191Z","iopub.execute_input":"2023-01-08T21:36:08.531161Z","iopub.status.idle":"2023-01-08T21:36:08.768354Z","shell.execute_reply.started":"2023-01-08T21:36:08.531108Z","shell.execute_reply":"2023-01-08T21:36:08.76737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Uncomment this cell to see that the TFrecord looks like.\n# raw_dataset = tf.data.TFRecordDataset(\"/kaggle/working/tfrecords/file_00-10000.tfrec\")\n# parsed_dataset = raw_dataset.map(parse_tfrecord_fn)\n\n# for features in parsed_dataset.take(1):\n#     for key in features.keys():\n#         if key != \"image\":\n#             print(f\"{key}: {features[key]}\")\n#         else:\n#             print(key, '-',type(key))\n\n#     print(f\"Image shape: {features['image'].shape}\")\n#     plt.figure(figsize=(7, 7))\n#     plt.imshow(features[\"image\"].numpy())\n#     plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}