{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Ash Color Images Dataset Creation Notebook\n\nCredit for original [Ash Color Dataset](https://www.kaggle.com/code/shashwatraman/contrails-dataset-ash-color).\n\n### Description Original:\nWe will create a Ash Color Images dataset of the satellite images of this competition using this notebook. Some main points:\n* Save only the labeled frame, which will be used for training.\n* Save only the human_pixel_masks.\n* Save the ash color image and the mask label in the same numpy file, so that we have to load only one file during training.\n* Save the final numpy arrays in float16 dtype to reduce total data size.\n\n### New contribution:\nWe also add band08, the reason is described in [Exploring Optimal Bands for Contrail Detection](https://www.kaggle.com/code/raki21/exploring-optimal-bands-for-contrail-detection).\nWe use a mean of human_individual masks for the train set instead of only the majority pixel mask. You can decide to train with these soft labels or just use mask=mask>0.51 to get back to the majority pixel mask. The Dataset for downloading is [Contrails Ash+band08+soft-label](https://www.kaggle.com/datasets/raki21/contrails-ash-band08-soft-label).","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport os\nfrom tqdm.notebook import tqdm\nfrom pathlib import Path","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-06T15:46:29.276548Z","iopub.execute_input":"2023-08-06T15:46:29.276916Z","iopub.status.idle":"2023-08-06T15:46:29.439405Z","shell.execute_reply.started":"2023-08-06T15:46:29.276887Z","shell.execute_reply":"2023-08-06T15:46:29.437482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '/kaggle/input/google-research-identify-contrails-reduce-global-warming/'","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:46:29.441874Z","iopub.execute_input":"2023-08-06T15:46:29.442268Z","iopub.status.idle":"2023-08-06T15:46:29.44805Z","shell.execute_reply.started":"2023-08-06T15:46:29.442226Z","shell.execute_reply":"2023-08-06T15:46:29.446795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Make the DataFrames\n\nWe will create train and valid dataframes, which will contain the record ids for each image.","metadata":{}},{"cell_type":"code","source":"train_rs = os.listdir(data_dir + 'train')\nvalid_rs = os.listdir(data_dir + 'validation')\n\ntrain_df = pd.DataFrame(train_rs, columns=['record_id'])\nvalid_df = pd.DataFrame(valid_rs, columns=['record_id'])\n\ntrain_df['train'] = 'train'\nvalid_df['train'] = 'valid'","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:46:29.449447Z","iopub.execute_input":"2023-08-06T15:46:29.449829Z","iopub.status.idle":"2023-08-06T15:46:30.267883Z","shell.execute_reply.started":"2023-08-06T15:46:29.449797Z","shell.execute_reply":"2023-08-06T15:46:30.266939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape, valid_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:46:30.270693Z","iopub.execute_input":"2023-08-06T15:46:30.27137Z","iopub.status.idle":"2023-08-06T15:46:30.278399Z","shell.execute_reply.started":"2023-08-06T15:46:30.271332Z","shell.execute_reply":"2023-08-06T15:46:30.27767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:46:30.279877Z","iopub.execute_input":"2023-08-06T15:46:30.280169Z","iopub.status.idle":"2023-08-06T15:46:30.321191Z","shell.execute_reply.started":"2023-08-06T15:46:30.280137Z","shell.execute_reply":"2023-08-06T15:46:30.319787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.to_csv('train_df.csv', index=False)\nvalid_df.to_csv('valid_df.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:46:30.32347Z","iopub.execute_input":"2023-08-06T15:46:30.324185Z","iopub.status.idle":"2023-08-06T15:46:30.372491Z","shell.execute_reply.started":"2023-08-06T15:46:30.324136Z","shell.execute_reply":"2023-08-06T15:46:30.370443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save the Images as Numpy arrays","metadata":{}},{"cell_type":"code","source":"def read_record(record_id, directory, mode):\n    record_data = {}\n    read = [\"band_08\", \"band_11\", \"band_14\", \"band_15\"]\n    if mode == 'train':\n        read.append('human_individual_masks')\n    elif mode == 'val':\n        read.append('human_pixel_masks')\n    for x in read:\n        if x == 'human_individual_masks':\n            individual = np.load(os.path.join(directory, record_id, x + \".npy\"))\n            record_data['human_pixel_masks'] = individual.sum(axis=3) / individual.shape[3]\n        else:\n            record_data[x] = np.load(os.path.join(directory, record_id, x + \".npy\"))\n    return record_data","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:46:30.374324Z","iopub.execute_input":"2023-08-06T15:46:30.375031Z","iopub.status.idle":"2023-08-06T15:46:30.383108Z","shell.execute_reply.started":"2023-08-06T15:46:30.374988Z","shell.execute_reply":"2023-08-06T15:46:30.381607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_T14_BOUNDS = (243, 303)                    # Changed from the original, just the name as it is the bounds for band14 not band11\n_CLOUD_TOP_TDIFF_BOUNDS = (-4, 5)\n_TDIFF_BOUNDS = (-4, 2)\nN_TIMES_BEFORE = 4\n\ndef normalize_range(data, bounds):\n    \"\"\"Maps data to the range [0, 1].\"\"\"\n    return (data - bounds[0]) / (bounds[1] - bounds[0])","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:46:30.385Z","iopub.execute_input":"2023-08-06T15:46:30.385424Z","iopub.status.idle":"2023-08-06T15:46:30.406117Z","shell.execute_reply.started":"2023-08-06T15:46:30.38539Z","shell.execute_reply":"2023-08-06T15:46:30.404589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This part is changed from the original, we add band08 normalized so its values are in similar range as r,g,b. \n# Here the original notebook clipped, which seems to be counterproductive as a lot of information is lost \n#and as far as I could discern it is not mentioned in the Ash Recipe\ndef get_false_color(record_data):\n    r = normalize_range(record_data[\"band_15\"] - record_data[\"band_14\"], _TDIFF_BOUNDS)\n    g = normalize_range(record_data[\"band_14\"] - record_data[\"band_11\"], _CLOUD_TOP_TDIFF_BOUNDS)\n    b = normalize_range(record_data[\"band_14\"], _T14_BOUNDS)\n    n = (record_data[\"band_08\"] - 230) / 20\n\n    false_color = np.stack([r, g, b, n], axis=2)               \n    img = false_color[..., N_TIMES_BEFORE]\n    \n    return img","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:46:30.407836Z","iopub.execute_input":"2023-08-06T15:46:30.408344Z","iopub.status.idle":"2023-08-06T15:46:30.430306Z","shell.execute_reply.started":"2023-08-06T15:46:30.408308Z","shell.execute_reply":"2023-08-06T15:46:30.429417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = Path('contrails')\npath.mkdir(exist_ok=True, parents=True)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:46:30.433777Z","iopub.execute_input":"2023-08-06T15:46:30.434211Z","iopub.status.idle":"2023-08-06T15:46:30.456707Z","shell.execute_reply.started":"2023-08-06T15:46:30.434177Z","shell.execute_reply":"2023-08-06T15:46:30.455001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Train\nfor i in tqdm(train_rs):\n    data = read_record(str(i), data_dir+'train', mode='train')\n    img = get_false_color(data)\n    final = np.dstack([img, data['human_pixel_masks']])\n    final = final.astype(np.float16)\n    \n    pathc = path/f\"{i}.npy\"\n    np.save(str(pathc), final)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:46:30.459996Z","iopub.execute_input":"2023-08-06T15:46:30.460833Z","iopub.status.idle":"2023-08-06T16:23:09.365781Z","shell.execute_reply.started":"2023-08-06T15:46:30.460777Z","shell.execute_reply":"2023-08-06T16:23:09.364219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Valid\nfor i in tqdm(valid_rs):\n    data = read_record(str(i), data_dir+'validation', mode='val')\n    img = get_false_color(data)\n    final = np.dstack([img, data['human_pixel_masks']])\n    final = final.astype(np.float16)\n    \n    pathc = path/f\"{i}.npy\"\n    np.save(str(pathc), final)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T16:23:09.374518Z","iopub.execute_input":"2023-08-06T16:23:09.37494Z","iopub.status.idle":"2023-08-06T16:26:19.650298Z","shell.execute_reply.started":"2023-08-06T16:23:09.374911Z","shell.execute_reply":"2023-08-06T16:26:19.648936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}