{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfrom glob import glob\nfrom tqdm import tqdm\nimport collections\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2023-06-25T16:44:44.734146Z","iopub.status.busy":"2023-06-25T16:44:44.734009Z","iopub.status.idle":"2023-06-25T16:44:45.23705Z","shell.execute_reply":"2023-06-25T16:44:45.236414Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"comp_data_dir = '/kaggle/input/google-research-identify-contrails-reduce-global-warming/'\ntrain_record_ids = os.listdir(f\"{comp_data_dir}/train/\")\nvalid_record_ids = os.listdir(f\"{comp_data_dir}/validation/\")\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_labelers_list = []\nfor train_record_id in tqdm(train_record_ids):\n    human_individual_masks = np.load(f\"{comp_data_dir}/train/{train_record_id}/human_individual_masks.npy\")\n    num_labelers = human_individual_masks.shape[3]\n    num_labelers_list.append(num_labelers)\n\nlabelers_unique_count = collections.Counter(num_labelers_list)\nsorted(labelers_unique_count.items())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ash-color/4labels","metadata":{}},{"cell_type":"code","source":"def normalize_range(data, bounds):\n    \"\"\"Maps data to the range [0, 1].\"\"\"\n    return (data - bounds[0]) / (bounds[1] - bounds[0])\n\n\ndef get_false_color(record_data):\n    _T11_BOUNDS = (243, 303)\n    _CLOUD_TOP_TDIFF_BOUNDS = (-4, 5)\n    _TDIFF_BOUNDS = (-4, 2)\n\n    r = normalize_range(record_data[\"band_15\"] - record_data[\"band_14\"], _TDIFF_BOUNDS)\n    g = normalize_range(record_data[\"band_14\"] - record_data[\"band_11\"], _CLOUD_TOP_TDIFF_BOUNDS)\n    b = normalize_range(record_data[\"band_14\"], _T11_BOUNDS)\n    images = np.clip(np.stack([r, g, b], axis=2), 0, 1)\n\n    return images\n\n\ndef read_record(record_id, comp_data_dir, mode):\n    record_data = {}\n    if mode in [\"train\"]:\n        bands_mask = [\"band_11\", \"band_14\", \"band_15\", \"human_individual_masks\"]\n    if mode in [\"validation\"]:\n        bands_mask = [\"band_11\", \"band_14\", \"band_15\", \"human_pixel_masks\"]\n    if mode in [\"test\"]:\n        bands_mask = [\"band_11\", \"band_14\", \"band_15\"]\n\n    for x in bands_mask:\n        record_data[x] = np.load(os.path.join(comp_data_dir, record_id, x + \".npy\"))\n    return record_data\n\n\ndef process_individual_masks(human_individual_masks):\n    num_masks = human_individual_masks.shape[-1]\n    masks_sum = np.zeros(human_individual_masks[..., 0].shape)\n    for i in range(num_masks):\n        masks_sum += human_individual_masks[..., i]\n    mask_agree = masks_sum/num_masks\n    mask = []\n    for p_agree in [0, 1/4, 2/4, 3/4]:\n        mask.append(np.where(mask_agree > p_agree, 1, 0))\n    mask = np.dstack(mask)\n    return mask\n\n\ndef create_dataset(comp_data_dir, dataset_dir, image_dir, label_dir, mode):\n    N_TIMES = 8\n    N_TIMES_LABELED = 4\n    os.makedirs(image_dir, exist_ok=True)\n    os.makedirs(label_dir, exist_ok=True)\n\n    input_dir = f\"{comp_data_dir}/{mode}\"\n    record_ids = os.listdir(input_dir)\n\n    df = pd.DataFrame(record_ids, columns=['record_id'])\n    df['image_path'] = image_dir + df['record_id'].astype(str) + f'_{N_TIMES_LABELED}.npy'\n    df['label_path'] = label_dir + df['record_id'].astype(str) + f'_{N_TIMES_LABELED}.npy'\n    df['time'] = N_TIMES_LABELED\n    df.to_csv(f\"{dataset_dir}/{mode}_df.csv\", index=False)\n\n    for record_id in tqdm(record_ids):\n        record_data = read_record(str(record_id), input_dir, mode)\n\n        images = get_false_color(record_data)\n        image = images[..., N_TIMES_LABELED]\n        image = image.astype(np.float16)\n        npy_image_path = f\"{image_dir}/{record_id}_{N_TIMES_LABELED}.npy\"\n        np.save(str(npy_image_path), image)\n\n        if mode in [\"train\",]:\n            label = process_individual_masks(record_data['human_individual_masks'])\n        if mode in [\"validation\",]:\n            label = record_data['human_pixel_masks']\n        if mode in [\"test\"]:\n            continue\n        label = label.astype(np.float16)\n        npy_label_path = f\"{label_dir}/{record_id}_{N_TIMES_LABELED}.npy\"\n        np.save(str(npy_label_path), label)","metadata":{"execution":{"iopub.execute_input":"2023-06-25T16:44:45.252384Z","iopub.status.busy":"2023-06-25T16:44:45.252176Z","iopub.status.idle":"2023-06-25T16:44:45.255901Z","shell.execute_reply":"2023-06-25T16:44:45.255508Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_train = \"/kaggle/working/dataset_train/ash_color_4labels/\"\ndataset_train_images = f\"{dataset_train}images/\"\ndataset_train_labels = f\"{dataset_train}labels/true/\"\ndataset_test = \"/kaggle/working/dataset_test/ash_color_4labels/images/\"\n\ncreate_dataset(comp_data_dir, dataset_train, dataset_train_images, dataset_train_labels, \"train\")\ncreate_dataset(comp_data_dir, dataset_train, dataset_train_images, dataset_train_labels, \"validation\")\n# create_dataset(data_dir, dataset_test, \"test\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, train_record_id in enumerate(tqdm(train_record_ids)):\n    dataset_label = np.load(f\"{dataset_train_labels}/{train_record_id}_4.npy\")\n    human_pixel_masks = np.load(f\"{comp_data_dir}/train/{train_record_id}/human_pixel_masks.npy\")\n\n    is_match = np.all(dataset_label[:, :, 2] == human_pixel_masks[:, :, 0])\n\n    if not is_match:\n        print(\"not matched\")\n        break\nprint(\"checked\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_train_dataset(i):\n    p_agree = [\"0\", \"1/4\", \"2/4\", \"3/4\"]\n    train_record_id = train_record_ids[i]\n    \n    label = np.load(f\"{dataset_train_labels}/{train_record_id}_4.npy\")\n    image = np.load(f\"{dataset_train_images}/{train_record_id}_4.npy\")\n    \n    fig, ax = plt.subplots(1, 5, figsize=(25, 20))\n    fig.tight_layout()\n    ax[0].imshow(image.astype(np.float32))\n    ax[0].axis('off')\n    for i in range(4):\n        ax[i+1].imshow(label[..., i])\n        ax[i+1].set_title(f\"p_agree: {str(p_agree[i])}\")\n        if p_agree[i] == \"2/4\":\n            ax[i+1].set_title(f\"p_agree: {str(p_agree[i])} (GT)\")\n        ax[i+1].axis('off')\n        \n\nplot_train_dataset(50)","metadata":{},"execution_count":null,"outputs":[]}]}