{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🤝 Intro","metadata":{}},{"cell_type":"markdown","source":"For the training notebook: https://www.kaggle.com/code/chazzer/neural-net-starter-g2net-train/","metadata":{}},{"cell_type":"markdown","source":"In 2015, gravitational waves were first detected via LIGO and Virgo.  \n~100 years after their theoretical postulation by Einstein in his General Theory of Relativity.  \nThe waves were detected thanks to a collision between to massive black holes.","metadata":{}},{"cell_type":"markdown","source":"![blackholes](https://i.gifer.com/Car.gif)","metadata":{}},{"cell_type":"markdown","source":"Today, we are trying to develop a machine learning model that can detect these waves, given some signal data.","metadata":{}},{"cell_type":"markdown","source":"Preface: this is a very rough cut and starter version of a potential convolutional neural network solution.  \nThe performance is very bad (extremely close to random guessing).  \nHowever, I hope you find some insights on how the model can be built.  \nPlease, if you have any feedback, let me know in the comments section!","metadata":{}},{"cell_type":"markdown","source":"The main limitation I've encoutered is on the memory side.  \nI've had to shrink down the signal via average pooling to stay within memory constraints.","metadata":{}},{"cell_type":"markdown","source":"# 🚚 Import","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"## Packages","metadata":{}},{"cell_type":"code","source":"import h5py\nfrom glob import glob\n\nimport numpy as np\nimport pandas as pd\n\nimport gc  # garbage collection\nimport json  # saving jsons\n\n# neural network\nimport tensorflow as tf\nfrom tensorflow.keras import layers\n\n# train test split\nfrom sklearn.model_selection import train_test_split\n\n# data transformation\nfrom cv2 import resize\n\n# visualization\nfrom tensorflow.keras.utils import plot_model\n\nfrom IPython.display import Image","metadata":{"execution":{"iopub.status.busy":"2022-10-09T18:22:19.676797Z","iopub.execute_input":"2022-10-09T18:22:19.677288Z","iopub.status.idle":"2022-10-09T18:22:19.918106Z","shell.execute_reply.started":"2022-10-09T18:22:19.677245Z","shell.execute_reply":"2022-10-09T18:22:19.916781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config = {\n    \"sample\": 0.01\n}","metadata":{"execution":{"iopub.status.busy":"2022-10-09T18:18:57.29927Z","iopub.execute_input":"2022-10-09T18:18:57.300025Z","iopub.status.idle":"2022-10-09T18:18:57.305611Z","shell.execute_reply.started":"2022-10-09T18:18:57.299983Z","shell.execute_reply":"2022-10-09T18:18:57.30451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('../input/neural-net-starter-g2net-train/extra_parameters.json', 'r') as f:\n    extra_parameters = json.load(f)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T18:19:09.377925Z","iopub.execute_input":"2022-10-09T18:19:09.378755Z","iopub.status.idle":"2022-10-09T18:19:09.387448Z","shell.execute_reply.started":"2022-10-09T18:19:09.378714Z","shell.execute_reply":"2022-10-09T18:19:09.38644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_shape = tuple(extra_parameters[\"input_shape\"])\nnorm_real = float(extra_parameters[\"norm_real\"])\nnorm_imag = float(extra_parameters[\"norm_imag\"])","metadata":{"execution":{"iopub.status.busy":"2022-10-09T18:20:38.335845Z","iopub.execute_input":"2022-10-09T18:20:38.336318Z","iopub.status.idle":"2022-10-09T18:20:38.342409Z","shell.execute_reply.started":"2022-10-09T18:20:38.336274Z","shell.execute_reply":"2022-10-09T18:20:38.34135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🧱 Build Model","metadata":{}},{"cell_type":"code","source":"class G2NetModel(tf.keras.Model):\n    def __init__(self):\n        super(G2NetModel, self).__init__()\n\n        self.conv1 = layers.Conv2D(32, (3,3), activation='relu')\n        self.dropout1 = layers.SpatialDropout2D(0.5)\n        self.pool1 = layers.MaxPool2D((2, 2))\n        self.conv2 = layers.Conv2D(64, (3,3), activation='relu')\n        self.pool2 = layers.MaxPool2D((2, 2))\n        self.conv3 = layers.Conv2D(128, (3,3), activation='relu')\n        self.pool3 = layers.MaxPool2D((2, 2))\n        self.flatten = layers.Flatten()\n        \n        self.dense1 = layers.Dense(1028, activation='relu')\n        self.batchnorm1 = layers.BatchNormalization()\n        self.dense2 = layers.Dense(32, activation='relu')\n        self.batchnorm2 = layers.BatchNormalization()\n        self.classifier = layers.Dense(1, activation='sigmoid')\n    \n    def call(self, inputs, training=False):\n        \n        # signal\n        x = self.conv1(inputs[0])\n        if training:\n            x = self.dropout1(x, training=training)\n        x = self.pool1(x)\n        x = self.conv2(x)\n        x = self.pool2(x)\n        x = self.conv3(x)\n        x = self.pool3(x)\n        x = self.flatten(x)\n        \n        x = self.dense1(x)\n        x = self.batchnorm1(x)\n        x = self.dense2(x)\n        x = self.batchnorm2(x)\n        \n        return self.classifier(x)\n    \n    def build_graph(self):\n        # helper method for plotting architecture\n        inputs = [\n            layers.Input(input_shape)\n        ]\n        return tf.keras.Model(\n            inputs=inputs,\n            outputs=self.call(inputs)\n        )","metadata":{"execution":{"iopub.status.busy":"2022-10-09T18:20:40.080516Z","iopub.execute_input":"2022-10-09T18:20:40.080937Z","iopub.status.idle":"2022-10-09T18:20:40.096739Z","shell.execute_reply.started":"2022-10-09T18:20:40.080897Z","shell.execute_reply":"2022-10-09T18:20:40.095507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = G2NetModel()\nmodel(\n    [\n        layers.Input(input_shape)\n    ]\n)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T18:20:40.662998Z","iopub.execute_input":"2022-10-09T18:20:40.663958Z","iopub.status.idle":"2022-10-09T18:20:41.105249Z","shell.execute_reply.started":"2022-10-09T18:20:40.663917Z","shell.execute_reply":"2022-10-09T18:20:41.104048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights('../input/neural-net-starter-g2net-train/g2net_starter_weights.h5')","metadata":{"execution":{"iopub.status.busy":"2022-10-09T18:20:42.754816Z","iopub.execute_input":"2022-10-09T18:20:42.755301Z","iopub.status.idle":"2022-10-09T18:20:43.962814Z","shell.execute_reply.started":"2022-10-09T18:20:42.755259Z","shell.execute_reply":"2022-10-09T18:20:43.961575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Image('../input/neural-net-starter-g2net-train/model.png')","metadata":{"execution":{"iopub.status.busy":"2022-10-09T18:20:43.964733Z","iopub.execute_input":"2022-10-09T18:20:43.965168Z","iopub.status.idle":"2022-10-09T18:20:43.979933Z","shell.execute_reply.started":"2022-10-09T18:20:43.96513Z","shell.execute_reply":"2022-10-09T18:20:43.978999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🏋️ Data ETL","metadata":{}},{"cell_type":"code","source":"def etl(filenames):\n    ret = []\n      \n    for filename in filenames:\n        with h5py.File(filename, \"r\") as f:\n            ind, *_ = f.keys()\n            \n            group1 = f[ind]\n            h1, _, freqs_key = group1.keys()\n\n            # still don't know how to handle frequencies in model\n            freq = np.array(group1[freqs_key])[:, None]\n\n            group2 = group1[h1]\n            signal_key, *_ = group2.keys()\n            signal = np.array(group2[signal_key])\n            \n            # we get the wave's real and imaginary parts\n            signal_real = np.real(signal)\n            signal_imag = np.imag(signal)\n            \n            signal_real = resize(signal_real, input_shape[:2])\n            signal_imag = resize(signal_imag, input_shape[:2])\n            \n            signal = np.stack([signal_real, signal_imag], axis=-1)\n            \n        \n        ret.append(\n            {\n                \"id\": ind,\n                \"signal\": signal,\n                \"freq\": freq\n            }\n        )\n    \n    # normalizing signal data between -1 and 1\n    for row in ret:\n        row[\"signal\"][:, :, 0] /= norm_real\n        row[\"signal\"][:, :, 1] /= norm_imag\n        \n    return pd.DataFrame(ret)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T18:23:08.365989Z","iopub.execute_input":"2022-10-09T18:23:08.366466Z","iopub.status.idle":"2022-10-09T18:23:08.378252Z","shell.execute_reply.started":"2022-10-09T18:23:08.366427Z","shell.execute_reply":"2022-10-09T18:23:08.37686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🧠 Batch Inference","metadata":{}},{"cell_type":"markdown","source":"Since the test data exceeds our RAM capabilities, we'll be doing our predictions in batches.","metadata":{}},{"cell_type":"code","source":"def batch(iterable, frac):\n    len_iterable = len(iterable)\n    batch_size = int(len_iterable * frac)\n    for b in range(0, len_iterable, batch_size):\n        yield iterable[b:min(b + batch_size, len_iterable)]","metadata":{"execution":{"iopub.status.busy":"2022-10-09T18:23:09.265147Z","iopub.execute_input":"2022-10-09T18:23:09.266631Z","iopub.status.idle":"2022-10-09T18:23:09.274451Z","shell.execute_reply.started":"2022-10-09T18:23:09.266537Z","shell.execute_reply":"2022-10-09T18:23:09.272792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_hdf5s = glob('../input/g2net-detecting-continuous-gravitational-waves/test/*')","metadata":{"execution":{"iopub.status.busy":"2022-10-09T18:23:09.501107Z","iopub.execute_input":"2022-10-09T18:23:09.501523Z","iopub.status.idle":"2022-10-09T18:23:09.534786Z","shell.execute_reply.started":"2022-10-09T18:23:09.501489Z","shell.execute_reply":"2022-10-09T18:23:09.533462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsub_df = pd.DataFrame(columns=[\"id\", \"target\"])\nfor i, test_batch in enumerate(batch(test_hdf5s, config[\"sample\"])):\n    test_data = etl(test_batch)\n    ids = test_data[\"id\"]\n    preds = model.predict([np.stack(test_data['signal'].values)])\n    \n    del test_data\n    gc.collect()\n    \n    batch_df = pd.DataFrame({\"id\": ids, \"target\": preds.T[0]})\n    \n    del ids\n    del preds\n    gc.collect()\n    \n    sub_df = pd.concat([sub_df, batch_df])\n    \n    del batch_df\n    gc.collect()\n    \n    print(f'Batch {i+1} done. {config[\"sample\"] * (i+1):.1%} completed')","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-10-09T18:23:10.147962Z","iopub.execute_input":"2022-10-09T18:23:10.148367Z","iopub.status.idle":"2022-10-09T18:23:44.247814Z","shell.execute_reply.started":"2022-10-09T18:23:10.148332Z","shell.execute_reply":"2022-10-09T18:23:44.246448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:04:01.374306Z","iopub.status.idle":"2022-10-06T12:04:01.374867Z","shell.execute_reply.started":"2022-10-06T12:04:01.374583Z","shell.execute_reply":"2022-10-06T12:04:01.374611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 💌 Submission","metadata":{}},{"cell_type":"code","source":"sub_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-06T12:04:01.382544Z","iopub.status.idle":"2022-10-06T12:04:01.383083Z","shell.execute_reply.started":"2022-10-06T12:04:01.382807Z","shell.execute_reply":"2022-10-06T12:04:01.382832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🙇 Conclusion","metadata":{}},{"cell_type":"markdown","source":"For model training, check out this notebook: https://www.kaggle.com/code/chazzer/neural-net-starter-g2net-train/","metadata":{}},{"cell_type":"markdown","source":"This is a very rough version of what could be a model.  \nThe results are very bad, close to random guessing.  \nHowever, we might improve these results, and I hope you can gain some insights from this notebook.  \nTake care!","metadata":{}}]}