{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Hello fellow Kagglers,\n\nThis notebook demonstrates the creation of a 360x360 dataset.\nThe general idea is to split the SFTs into a real and imaginary frame and split the frames into 360x360 patches.\nA 360x720 SFT will be split into 2 360x360 patches and a 360x1080 SFT will be split into 3 360x360 patches.\nThe original SFTs are truncated to a multiple of 360, thus a 360x750 SFT will be split into 2 360x306 patches, dropping the remaining 360x30 patch.\n\nSince each recording consists of two locations, H (Hanford) and L (Livingston), and each SFT is split into a real and imaginary part, each timestamp has 4 \"channels\".\nThese channels can be fed into a CNN in multuple ways, notebook coming soon.\n\nAll complex128 training samples are dropped as they are inconsistent with the majority of complex64 samples in terms of length and values.\n\nThis is a work in progress, expect some new findings soon!","metadata":{}},{"cell_type":"code","source":"# Package to read HDF5 files\n!pip install -q h5py","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:19:52.427859Z","iopub.execute_input":"2022-10-06T11:19:52.428506Z","iopub.status.idle":"2022-10-06T11:20:05.935689Z","shell.execute_reply.started":"2022-10-06T11:19:52.428379Z","shell.execute_reply":"2022-10-06T11:20:05.934536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport tensorflow as tf\n\nfrom tqdm.notebook import tqdm\n\nimport h5py\nimport glob\nimport cv2\nimport gc\nimport os\nimport sys\n\ntqdm.pandas()","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:05.939173Z","iopub.execute_input":"2022-10-06T11:20:05.939617Z","iopub.status.idle":"2022-10-06T11:20:12.298078Z","shell.execute_reply.started":"2022-10-06T11:20:05.939568Z","shell.execute_reply":"2022-10-06T11:20:12.297038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# MatplotLib Configuration","metadata":{}},{"cell_type":"code","source":"# MatplotLib Global Settings\nmpl.rcParams.update(mpl.rcParamsDefault)\nmpl.rcParams['xtick.labelsize'] = 16\nmpl.rcParams['ytick.labelsize'] = 16\nmpl.rcParams['axes.labelsize'] = 18\nmpl.rcParams['axes.titlesize'] = 24","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:12.299729Z","iopub.execute_input":"2022-10-06T11:20:12.300715Z","iopub.status.idle":"2022-10-06T11:20:12.308464Z","shell.execute_reply.started":"2022-10-06T11:20:12.300672Z","shell.execute_reply":"2022-10-06T11:20:12.307053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read Train DataFrame","metadata":{}},{"cell_type":"code","source":"# Read Train DataFrame\ntrain_labels = pd.read_csv('/kaggle/input/g2net-detecting-continuous-gravitational-waves/train_labels.csv')\n\ndisplay(train_labels.info())\n\ndisplay(train_labels.head())","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:12.310249Z","iopub.execute_input":"2022-10-06T11:20:12.311688Z","iopub.status.idle":"2022-10-06T11:20:12.389232Z","shell.execute_reply.started":"2022-10-06T11:20:12.311639Z","shell.execute_reply":"2022-10-06T11:20:12.387854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"# Number of Samples in train dataset\nN_SAMPLES = len(train_labels)\n# Make 360x360 Patches\nTARGET_HEIGHT = 360\nTARGET_WIDTH = 360\nprint(f'TARGET_HEIGHT: {TARGET_HEIGHT}, TARGET_WIDTH: {TARGET_WIDTH}')\n\nIS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\n\nTRAIN_DIR = '/kaggle/input/g2net-detecting-continuous-gravitational-waves/train'\n\nINPUTS = ['x_h_r', 'x_h_i', 'x_l_r', 'x_l_i']","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:12.392126Z","iopub.execute_input":"2022-10-06T11:20:12.392521Z","iopub.status.idle":"2022-10-06T11:20:12.399827Z","shell.execute_reply.started":"2022-10-06T11:20:12.392488Z","shell.execute_reply":"2022-10-06T11:20:12.398809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Target Distribution","metadata":{}},{"cell_type":"code","source":"# Target distribution, slightly biased towards 1\nplt.figure(figsize=(8, 8))\ntrain_labels['target'].value_counts().plot(kind='pie', autopct='%1.1f%%', title='Target Distribution')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:12.401369Z","iopub.execute_input":"2022-10-06T11:20:12.402032Z","iopub.status.idle":"2022-10-06T11:20:12.667712Z","shell.execute_reply.started":"2022-10-06T11:20:12.401995Z","shell.execute_reply":"2022-10-06T11:20:12.666864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Data Type","metadata":{}},{"cell_type":"code","source":"# Get recording data type\n# A handful of complex128 recordings are present which will be ignored\ndef get_dtype(train_id):\n    file = h5py.File(f'{TRAIN_DIR}/{train_id}.hdf5', 'r')[train_id]\n    return file['H1']['SFTs'].dtype\n\ntrain_labels['dtype'] = train_labels['id'].progress_apply(get_dtype)","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:12.669215Z","iopub.execute_input":"2022-10-06T11:20:12.669815Z","iopub.status.idle":"2022-10-06T11:20:12.964876Z","shell.execute_reply.started":"2022-10-06T11:20:12.669777Z","shell.execute_reply":"2022-10-06T11:20:12.964043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_FLOAT64_SAMPLES = (train_labels['dtype'] == 'complex64').sum()\nprint(f'N_FLOAT64_SAMPLES: {N_FLOAT64_SAMPLES}')\n\ndisplay(train_labels['dtype'].value_counts().to_frame())","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:12.965907Z","iopub.execute_input":"2022-10-06T11:20:12.966827Z","iopub.status.idle":"2022-10-06T11:20:12.979357Z","shell.execute_reply.started":"2022-10-06T11:20:12.966792Z","shell.execute_reply":"2022-10-06T11:20:12.978345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Dimensions","metadata":{}},{"cell_type":"code","source":"train_dimension_rows = []\n\nfor train_id in tqdm(train_labels['id']):       \n    file = h5py.File(f'{TRAIN_DIR}/{train_id}.hdf5', 'r')[train_id]\n    SFT_H = file['H1']['SFTs']\n    SFT_L = file['L1']['SFTs']\n    train_dimension_rows.append({\n        'id': train_id,\n        'H_height': SFT_H.shape[0],\n        'H_width': SFT_H.shape[1],\n        'L_height': SFT_L.shape[0],\n        'L_width': SFT_L.shape[1],\n    })        ","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:12.980573Z","iopub.execute_input":"2022-10-06T11:20:12.981135Z","iopub.status.idle":"2022-10-06T11:20:13.233621Z","shell.execute_reply.started":"2022-10-06T11:20:12.9811Z","shell.execute_reply":"2022-10-06T11:20:13.232364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = train_labels.merge(pd.DataFrame(train_dimension_rows), on='id')\n\ndisplay(train_labels.head())","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:13.235538Z","iopub.execute_input":"2022-10-06T11:20:13.236815Z","iopub.status.idle":"2022-10-06T11:20:13.272902Z","shell.execute_reply.started":"2022-10-06T11:20:13.236767Z","shell.execute_reply":"2022-10-06T11:20:13.271525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Height is consistantly 360\ntrain_labels[['H_height', 'L_height']].value_counts().to_frame(name='Count')","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:13.274612Z","iopub.execute_input":"2022-10-06T11:20:13.274986Z","iopub.status.idle":"2022-10-06T11:20:13.291728Z","shell.execute_reply.started":"2022-10-06T11:20:13.27495Z","shell.execute_reply":"2022-10-06T11:20:13.289954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Width is ~4500, except the 3 complex128 recordings\nplt.figure(figsize=(8, 5))\nplt.title('Handford With Distribution')\ntrain_labels['H_width'].plot(kind='hist', bins=100)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:13.294304Z","iopub.execute_input":"2022-10-06T11:20:13.295494Z","iopub.status.idle":"2022-10-06T11:20:13.68747Z","shell.execute_reply.started":"2022-10-06T11:20:13.295439Z","shell.execute_reply":"2022-10-06T11:20:13.686231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Width is ~4500, except the 3 complex128 recordings\nplt.figure(figsize=(8, 5))\nplt.title('Livingston Width Distribution')\ntrain_labels['L_width'].plot(kind='hist', bins=100)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:13.688687Z","iopub.execute_input":"2022-10-06T11:20:13.68901Z","iopub.status.idle":"2022-10-06T11:20:14.233468Z","shell.execute_reply.started":"2022-10-06T11:20:13.688981Z","shell.execute_reply":"2022-10-06T11:20:14.232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Dimensions","metadata":{}},{"cell_type":"code","source":"# Create target directories\n!rm -rf train_samples\n!mkdir -p train_samples/{x,target}\n!ls -l train_samples","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:14.238675Z","iopub.execute_input":"2022-10-06T11:20:14.23911Z","iopub.status.idle":"2022-10-06T11:20:17.549893Z","shell.execute_reply.started":"2022-10-06T11:20:14.239065Z","shell.execute_reply":"2022-10-06T11:20:17.548371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This large function actually generates the 360x360 patches\ndef get_train_stats():\n    c = 0\n    # Loop over all training samples\n    for row_idx, row in tqdm(train_labels.iterrows(), total=N_SAMPLES):        \n        train_id = row['id']\n        # Skip non-complex64 samples\n        if row['dtype'] != 'complex64':\n            continue\n            \n        # Read SFTs as numpy arrays\n        with h5py.File(f'{TRAIN_DIR}/{train_id}.hdf5', 'r') as file:\n            SFT_H = np.array(file[train_id]['H1']['SFTs'])\n            SFT_L = np.array(file[train_id]['L1']['SFTs'])\n        \n        # Split into real and imaginary part\n        SFT_H_SPLIT = SFT_H.view(np.float32).reshape([*SFT_H.shape, 2])\n        SFT_L_SPLIT = SFT_L.view(np.float32).reshape([*SFT_L.shape, 2])\n        # Transpose to get channel(real/imaginary) first\n        SFT_H_SPLIT = np.transpose(SFT_H_SPLIT, [2,0,1])\n        SFT_L_SPLIT = np.transpose(SFT_L_SPLIT, [2,0,1])\n        \n        # Create target array\n        N = min(row['H_width'], row['L_width']) // TARGET_HEIGHT\n        x = np.zeros(shape=[N, len(INPUTS), TARGET_HEIGHT, TARGET_HEIGHT], dtype=np.float32)\n        # Get patches\n        for offset in range(N):\n            x[offset, 0] = SFT_H_SPLIT[0, :, offset * TARGET_HEIGHT:(offset + 1) * TARGET_HEIGHT]\n            x[offset, 1] = SFT_H_SPLIT[1, :, offset * TARGET_HEIGHT:(offset + 1) * TARGET_HEIGHT]\n            x[offset, 2] = SFT_L_SPLIT[0, :, offset * TARGET_HEIGHT:(offset + 1) * TARGET_HEIGHT]\n            x[offset, 3] = SFT_L_SPLIT[1,:, offset * TARGET_HEIGHT:(offset + 1) * TARGET_HEIGHT]\n        \n        # Save patches and target\n        np.save(f'./train_samples/x/{c}.npy', x)\n        np.save(f'./train_samples/target/{c}.npy', np.array(row['target']))\n        c += 1\n    \n    return c\n\nN_TRAIN_SAMPLES = get_train_stats()\nprint(f'N_TRAIN_SAMPLES: {N_TRAIN_SAMPLES}')","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:17.552898Z","iopub.execute_input":"2022-10-06T11:20:17.553458Z","iopub.status.idle":"2022-10-06T11:20:26.285034Z","shell.execute_reply.started":"2022-10-06T11:20:17.553399Z","shell.execute_reply":"2022-10-06T11:20:26.283725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Normalize","metadata":{}},{"cell_type":"code","source":"# Mean/STD target array\nX_STATS = {\n    'x_h_r_mean': 0.0, 'x_h_r_std': 0.0,\n    'x_h_i_mean': 0.0, 'x_h_i_std': 0.0,\n    'x_l_r_mean': 0.0, 'x_l_r_std': 0.0,\n    'x_l_i_mean': 0.0, 'x_l_i_std': 0.0,\n}","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:26.286636Z","iopub.execute_input":"2022-10-06T11:20:26.286985Z","iopub.status.idle":"2022-10-06T11:20:26.292205Z","shell.execute_reply.started":"2022-10-06T11:20:26.286953Z","shell.execute_reply":"2022-10-06T11:20:26.290843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute mean/STD\nfor i in tqdm(range(N_TRAIN_SAMPLES)):\n    v = np.load(f'./train_samples/x/{i}.npy')\n    for k_idx, k in enumerate(INPUTS):\n        X_STATS[f'{k}_mean'] += v[k_idx].mean() / N_TRAIN_SAMPLES\n        X_STATS[f'{k}_std'] += v[k_idx].std() / N_TRAIN_SAMPLES","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:26.293653Z","iopub.execute_input":"2022-10-06T11:20:26.294761Z","iopub.status.idle":"2022-10-06T11:20:26.663066Z","shell.execute_reply.started":"2022-10-06T11:20:26.294723Z","shell.execute_reply":"2022-10-06T11:20:26.66186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Values are extremely tiny\n# A neural network will get NaN values when training on these small values\ndisplay(pd.Series(X_STATS).to_frame('Value'))","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:26.664773Z","iopub.execute_input":"2022-10-06T11:20:26.665274Z","iopub.status.idle":"2022-10-06T11:20:26.681324Z","shell.execute_reply.started":"2022-10-06T11:20:26.665225Z","shell.execute_reply":"2022-10-06T11:20:26.679819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Normalize Images\n\nNormalize to mean 0 and std 1 for neural network input","metadata":{}},{"cell_type":"code","source":"# normalize input to mean 0 std 1\nfor i in tqdm(range(N_TRAIN_SAMPLES)):\n    # Load Image\n    v = np.load(f'./train_samples/x/{i}.npy')\n    \n    # Normalise Image\n    for k_idx, k in enumerate(INPUTS):\n        v[:,k_idx] = (v[:,k_idx] - X_STATS[f'{k}_mean']) / X_STATS[f'{k}_std']\n    \n    # Save Normalised Images\n    np.save(f'./train_samples/x/{i}.npy', v)","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:26.683614Z","iopub.execute_input":"2022-10-06T11:20:26.684033Z","iopub.status.idle":"2022-10-06T11:20:27.159956Z","shell.execute_reply.started":"2022-10-06T11:20:26.683968Z","shell.execute_reply":"2022-10-06T11:20:27.158704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Data Examples\n\nVisualization of the training samples, can not make much of this, a neural network will have to figure out how to deduce the target...","metadata":{}},{"cell_type":"code","source":"# Number of Samples\nN = 10\n# Figure\nfig, axes = plt.subplots(nrows=N, ncols=4, figsize=(25, 6 * N))\n\nfor r_idx, r in enumerate(np.random.choice(N_TRAIN_SAMPLES, N)):\n    # Load Patches\n    imagess = np.load(f'./train_samples/x/{r}.npy')\n    target = np.load(f'./train_samples/target/{r}.npy')\n    # Select Random Patch\n    images = imagess[np.random.choice(len(imagess), 1).squeeze()]\n    for k_idx, k in enumerate(INPUTS):\n        img = images[k_idx]\n        axes[r_idx, k_idx].imshow(img)\n        axes[r_idx, k_idx].set_title(\n            f'Img {r} {k} [{target}], mean: {img.mean():.2E}, std: {img.std():.2f}',\n            size=16,\n            pad=16\n        )\n        axes[r_idx, k_idx].tick_params(axis='x', labelsize=12)\n        axes[r_idx, k_idx].tick_params(axis='y', labelsize=12)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-06T11:20:27.161745Z","iopub.execute_input":"2022-10-06T11:20:27.162101Z","iopub.status.idle":"2022-10-06T11:20:36.969182Z","shell.execute_reply.started":"2022-10-06T11:20:27.162067Z","shell.execute_reply":"2022-10-06T11:20:36.966358Z"},"trusted":true},"execution_count":null,"outputs":[]}]}