{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":51753,"databundleVersionId":5692552,"sourceType":"competition"}],"dockerImageVersionId":30497,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Ash Color Images Dataset Creation Notebook\n\nWe will create a Ash Color Images dataset of the satellite images of this competition for our models using this notebook. Some main points:\n* Save only the labeled frame, which will be used for training.\n* Save only the human_pixel_masks.\n* Save the ash color image and the mask label in the same numpy file, so that we have to load only one file during training.\n* Save the final numpy arrays in float16 dtype to reduce total data size.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport os\nfrom tqdm.notebook import tqdm\nfrom pathlib import Path","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-04T06:50:14.191404Z","iopub.execute_input":"2023-06-04T06:50:14.192026Z","iopub.status.idle":"2023-06-04T06:50:14.400493Z","shell.execute_reply.started":"2023-06-04T06:50:14.191981Z","shell.execute_reply":"2023-06-04T06:50:14.398738Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dir = '/kaggle/input/google-research-identify-contrails-reduce-global-warming/'","metadata":{"execution":{"iopub.status.busy":"2023-06-04T06:50:14.402667Z","iopub.execute_input":"2023-06-04T06:50:14.403339Z","iopub.status.idle":"2023-06-04T06:50:14.409312Z","shell.execute_reply.started":"2023-06-04T06:50:14.403301Z","shell.execute_reply":"2023-06-04T06:50:14.407406Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Make the DataFrames\n\nWe will create train and valid dataframes, which will contain the record ids for each image.","metadata":{}},{"cell_type":"code","source":"train_rs = os.listdir(data_dir + 'train')\nvalid_rs = os.listdir(data_dir + 'validation')\n\ntrain_df = pd.DataFrame(train_rs, columns=['record_id'])\nvalid_df = pd.DataFrame(valid_rs, columns=['record_id'])\n\ntrain_df['train'] = 'train'\nvalid_df['train'] = 'valid'","metadata":{"execution":{"iopub.status.busy":"2023-06-04T06:50:14.743934Z","iopub.execute_input":"2023-06-04T06:50:14.745062Z","iopub.status.idle":"2023-06-04T06:50:15.830865Z","shell.execute_reply.started":"2023-06-04T06:50:14.745021Z","shell.execute_reply":"2023-06-04T06:50:15.829824Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape, valid_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-04T06:50:15.835456Z","iopub.execute_input":"2023-06-04T06:50:15.836521Z","iopub.status.idle":"2023-06-04T06:50:15.848269Z","shell.execute_reply.started":"2023-06-04T06:50:15.836473Z","shell.execute_reply":"2023-06-04T06:50:15.845947Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T06:50:15.85031Z","iopub.execute_input":"2023-06-04T06:50:15.850829Z","iopub.status.idle":"2023-06-04T06:50:15.886012Z","shell.execute_reply.started":"2023-06-04T06:50:15.850784Z","shell.execute_reply":"2023-06-04T06:50:15.884089Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.to_csv('train_df.csv', index=False)\nvalid_df.to_csv('valid_df.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T06:50:16.001051Z","iopub.execute_input":"2023-06-04T06:50:16.001972Z","iopub.status.idle":"2023-06-04T06:50:16.095957Z","shell.execute_reply.started":"2023-06-04T06:50:16.001934Z","shell.execute_reply":"2023-06-04T06:50:16.094715Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Save the Images as Numpy arrays","metadata":{}},{"cell_type":"code","source":"def read_record(record_id, directory):\n    record_data = {}\n    for x in [\n        \"band_11\", \n        \"band_14\", \n        \"band_15\", \n        \"human_pixel_masks\"\n    ]:\n\n        record_data[x] = np.load(os.path.join(directory, record_id, x + \".npy\"))\n    \n    return record_data","metadata":{"execution":{"iopub.status.busy":"2023-06-04T06:50:17.199171Z","iopub.execute_input":"2023-06-04T06:50:17.19956Z","iopub.status.idle":"2023-06-04T06:50:17.205516Z","shell.execute_reply.started":"2023-06-04T06:50:17.199529Z","shell.execute_reply":"2023-06-04T06:50:17.204489Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"_T11_BOUNDS = (243, 303)\n_CLOUD_TOP_TDIFF_BOUNDS = (-4, 5)\n_TDIFF_BOUNDS = (-4, 2)\n\ndef normalize_range(data, bounds):\n    \"\"\"Maps data to the range [0, 1].\"\"\"\n    return (data - bounds[0]) / (bounds[1] - bounds[0])\n\nN_TIMES_BEFORE = 4","metadata":{"execution":{"iopub.status.busy":"2023-06-04T06:50:17.758928Z","iopub.execute_input":"2023-06-04T06:50:17.759762Z","iopub.status.idle":"2023-06-04T06:50:17.765541Z","shell.execute_reply.started":"2023-06-04T06:50:17.759726Z","shell.execute_reply":"2023-06-04T06:50:17.764645Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_false_color(record_data):\n    _T11_BOUNDS = (243, 303)\n    _CLOUD_TOP_TDIFF_BOUNDS = (-4, 5)\n    _TDIFF_BOUNDS = (-4, 2)\n\n    r = normalize_range(record_data[\"band_15\"] - record_data[\"band_14\"], _TDIFF_BOUNDS)\n    g = normalize_range(record_data[\"band_14\"] - record_data[\"band_11\"], _CLOUD_TOP_TDIFF_BOUNDS)\n    b = normalize_range(record_data[\"band_14\"], _T11_BOUNDS)\n    false_color = np.clip(np.stack([r, g, b], axis=2), 0, 1)\n    img = false_color[..., N_TIMES_BEFORE]\n    \n    return img","metadata":{"execution":{"iopub.status.busy":"2023-06-04T06:50:18.29709Z","iopub.execute_input":"2023-06-04T06:50:18.298264Z","iopub.status.idle":"2023-06-04T06:50:18.306259Z","shell.execute_reply.started":"2023-06-04T06:50:18.298212Z","shell.execute_reply":"2023-06-04T06:50:18.304702Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = Path('contrails')\npath.mkdir(exist_ok=True, parents=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T06:50:19.256873Z","iopub.execute_input":"2023-06-04T06:50:19.257276Z","iopub.status.idle":"2023-06-04T06:50:19.263111Z","shell.execute_reply.started":"2023-06-04T06:50:19.257243Z","shell.execute_reply":"2023-06-04T06:50:19.261886Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Train\nfor i in tqdm(train_rs):\n    data = read_record(str(i), data_dir+'train')\n    img = get_false_color(data)\n    final = np.dstack([img, data['human_pixel_masks']])\n    final = final.astype(np.float16)\n    \n    pathc = path/f\"{i}.npy\"\n    np.save(str(pathc), final)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T08:11:34.128057Z","iopub.execute_input":"2023-05-31T08:11:34.128421Z","iopub.status.idle":"2023-05-31T08:11:34.136764Z","shell.execute_reply.started":"2023-05-31T08:11:34.128393Z","shell.execute_reply":"2023-05-31T08:11:34.13586Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Valid\nfor i in tqdm(valid_rs):\n    data = read_record(str(i), data_dir+'validation')\n    img = get_false_color(data)\n    final = np.dstack([img, data['human_pixel_masks']])\n    final = final.astype(np.float16)\n    \n    pathc = path/f\"{i}.npy\"\n    np.save(str(pathc), final)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T08:12:27.737322Z","iopub.execute_input":"2023-05-31T08:12:27.737787Z"},"trusted":true},"outputs":[],"execution_count":null}]}