{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom path import Path\nimport json\nimport os\nimport pprint","metadata":{"execution":{"iopub.status.busy":"2023-05-11T17:55:09.52893Z","iopub.execute_input":"2023-05-11T17:55:09.529365Z","iopub.status.idle":"2023-05-11T17:55:09.53553Z","shell.execute_reply.started":"2023-05-11T17:55:09.529332Z","shell.execute_reply":"2023-05-11T17:55:09.534383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_PATH = Path(\"/kaggle/input/google-research-identify-contrails-reduce-global-warming\")\n\n# Metadata extraction\nwith open(DATA_PATH / \"train_metadata.json\") as f:\n    train_metadata = json.load(f)","metadata":{"execution":{"iopub.status.busy":"2023-05-11T17:52:29.214645Z","iopub.execute_input":"2023-05-11T17:52:29.215029Z","iopub.status.idle":"2023-05-11T17:52:29.68641Z","shell.execute_reply.started":"2023-05-11T17:52:29.214999Z","shell.execute_reply":"2023-05-11T17:52:29.684974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train metadata sample\nfor k in train_metadata[0]:\n    print(f\"\\n{k}: ({type(train_metadata[0][k])}) \\n{train_metadata[0][k]}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-11T18:01:07.273027Z","iopub.execute_input":"2023-05-11T18:01:07.273464Z","iopub.status.idle":"2023-05-11T18:01:07.280756Z","shell.execute_reply.started":"2023-05-11T18:01:07.273428Z","shell.execute_reply":"2023-05-11T18:01:07.279151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the number of samples for each set \n\ntrain_ids = list(map(int, os.listdir(DATA_PATH / \"train\")))\nval_ids = list(map(int, os.listdir(DATA_PATH / \"validation\")))\ntest_ids = list(map(int, os.listdir(DATA_PATH / \"test\")))\n\nprint(\"Train samples:\", len(train_ids))\nprint(\"Validation samples:\", len(val_ids))\nprint(\"Test samples:\", len(test_ids))","metadata":{"execution":{"iopub.status.busy":"2023-05-10T22:45:25.494918Z","iopub.execute_input":"2023-05-10T22:45:25.495368Z","iopub.status.idle":"2023-05-10T22:45:25.515515Z","shell.execute_reply.started":"2023-05-10T22:45:25.495326Z","shell.execute_reply":"2023-05-10T22:45:25.514331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking that all samples contain the same layers and same image size\n\n# Checking Train set\ntrain_example_files = os.listdir(DATA_PATH / \"train\" / str(train_ids[0]))\nprint(\"\\nTrain exmaple:\", train_example_files)\n\nfor id in train_ids:\n    sample_files = os.listdir(DATA_PATH / \"train\" / str(id))\n    if sample_files != train_example_files:\n        print(\"\\nDifferent train sample:\", id)\n        print(sample_files)\n        \n# Checking Validation set\nval_example_files = os.listdir(DATA_PATH / \"validation\" / str(val_ids[0]))\nprint(\"\\nValidation example:\", val_example_files)\n\nfor id in val_ids:\n    sample_files = os.listdir(DATA_PATH / \"validation\" / str(id))\n    if sample_files != val_example_files:\n        print(\"\\nDifferent validation sample:\", id)\n        print(sample_files)","metadata":{"execution":{"iopub.status.busy":"2023-05-10T22:45:25.602584Z","iopub.execute_input":"2023-05-10T22:45:25.602955Z","iopub.status.idle":"2023-05-10T22:45:36.473803Z","shell.execute_reply.started":"2023-05-10T22:45:25.602924Z","shell.execute_reply":"2023-05-10T22:45:36.472792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So apparently all train files have the same files and also all validation files have the same files, between they both only differ in the \"human_individual_masks.npy\" which is in the train set but not in the validation one. \n\nThe test set only contains the bands 08 to 16 and no masks.","metadata":{}},{"cell_type":"code","source":"# Checking that all samples have the same dimensions\n\nfor file in os.listdir(DATA_PATH / \"train\" / str(train_ids[0])):\n    img = np.load(DATA_PATH / \"train\" / str(train_ids[0]) / file)\n    print(img.shape, f\"  \\t{file}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-10T22:45:36.475743Z","iopub.execute_input":"2023-05-10T22:45:36.476364Z","iopub.status.idle":"2023-05-10T22:45:36.507698Z","shell.execute_reply.started":"2023-05-10T22:45:36.476332Z","shell.execute_reply":"2023-05-10T22:45:36.506754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_sample_images(sample, data_set=\"train\"):\n    rows = 9 \n    if data_set == \"train\":\n        rows = 11\n    elif data_set == \"validation\":\n        rows = 10\n        \n    fig, axes = plt.subplots(rows, 8, figsize=(16, 20))\n    fig.tight_layout(pad=0.0)\n    \n    i = 0\n    for img_path in os.listdir(DATA_PATH / data_set / str(sample)):\n        if img_path[:4] == \"band\":\n            img = np.load(DATA_PATH / data_set / str(sample) / img_path)\n            for l in range(8):\n                ax = axes[i, l]\n                layer = img[:, :, l]\n                im = ax.imshow(layer)\n                ax.axis(\"off\")\n                if i == 0:\n                    ax.set_title(f\"Layer {l + 1}\")\n            i += 1\n            \n    if data_set == \"train\":\n        img = np.load(DATA_PATH / data_set / str(sample) / \"human_individual_masks.npy\")\n        for l in range(8):\n            ax = axes[9, l]\n            if l < 4:\n                layer = img[:, :, :, l]\n                im = ax.imshow(layer)\n            ax.axis(\"off\")\n            \n        img = np.load(DATA_PATH / data_set / str(sample) / \"human_pixel_masks.npy\")\n        for l in range(8):\n            ax = axes[10, l]\n            if l < 1:\n                im = ax.imshow(img)\n            ax.axis(\"off\")\n            \n    if data_set == \"validation\":\n        img = np.load(DATA_PATH / data_set / str(sample) / \"human_pixel_masks.npy\")\n        for l in range(8):\n            ax = axes[9, l]\n            if l < 1:\n                im = ax.imshow(img)\n            ax.axis(\"off\")\n    \n    plt.tight_layout()\n    plt.show()\n    \n    \nplot_sample_images(train_ids[0])","metadata":{"execution":{"iopub.status.busy":"2023-05-10T22:48:11.065616Z","iopub.execute_input":"2023-05-10T22:48:11.066005Z","iopub.status.idle":"2023-05-10T22:48:17.452641Z","shell.execute_reply.started":"2023-05-10T22:48:11.065972Z","shell.execute_reply":"2023-05-10T22:48:17.450449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_sample_images(train_ids[1])","metadata":{"execution":{"iopub.status.busy":"2023-05-10T22:48:17.455094Z","iopub.execute_input":"2023-05-10T22:48:17.455454Z","iopub.status.idle":"2023-05-10T22:48:24.212569Z","shell.execute_reply.started":"2023-05-10T22:48:17.455422Z","shell.execute_reply":"2023-05-10T22:48:24.210591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_sample_images(val_ids[0], \"validation\")","metadata":{"execution":{"iopub.status.busy":"2023-05-10T22:48:24.214679Z","iopub.execute_input":"2023-05-10T22:48:24.215447Z","iopub.status.idle":"2023-05-10T22:48:30.726217Z","shell.execute_reply.started":"2023-05-10T22:48:24.215407Z","shell.execute_reply":"2023-05-10T22:48:30.724909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_sample_images(val_ids[1], \"validation\")","metadata":{"execution":{"iopub.status.busy":"2023-05-10T22:48:30.728644Z","iopub.execute_input":"2023-05-10T22:48:30.729463Z","iopub.status.idle":"2023-05-10T22:48:37.954851Z","shell.execute_reply.started":"2023-05-10T22:48:30.72942Z","shell.execute_reply":"2023-05-10T22:48:37.953252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_sample_images(test_ids[0], \"test\")","metadata":{"execution":{"iopub.status.busy":"2023-05-10T22:48:37.95673Z","iopub.execute_input":"2023-05-10T22:48:37.957072Z","iopub.status.idle":"2023-05-10T22:48:44.981655Z","shell.execute_reply.started":"2023-05-10T22:48:37.95704Z","shell.execute_reply":"2023-05-10T22:48:44.979052Z"},"trusted":true},"execution_count":null,"outputs":[]}]}