{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# G2Net Create TFR Spectrogram Datasets\n\n## references:\n\n## [How To Create TFRecords](https://www.kaggle.com/cdeotte/how-to-create-tfrecords)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport time\nimport h5py\nimport tensorflow as tf\nimport cv2\n\nfrom tqdm.auto import tqdm\nfrom sklearn.model_selection import StratifiedKFold\n\n# Train metadata\ndi = '/kaggle/input/g2net-detecting-continuous-gravitational-waves'\ndf = pd.read_csv(di + '/train_labels.csv')\ndf = df[df.target >= 0]  # Remove 3 unknowns (target = -1)\ndf = df.reset_index()\nlen(df)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:35:06.604988Z","iopub.execute_input":"2022-10-28T05:35:06.605374Z","iopub.status.idle":"2022-10-28T05:35:06.618258Z","shell.execute_reply.started":"2022-10-28T05:35:06.605343Z","shell.execute_reply":"2022-10-28T05:35:06.616806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FOLDS = 5\nSEED = 42","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:35:06.620521Z","iopub.execute_input":"2022-10-28T05:35:06.620907Z","iopub.status.idle":"2022-10-28T05:35:06.627529Z","shell.execute_reply.started":"2022-10-28T05:35:06.620862Z","shell.execute_reply":"2022-10-28T05:35:06.62633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kfold = StratifiedKFold(n_splits=FOLDS, random_state=SEED, shuffle=True)\nfor i, (train_idx, valid_idx) in enumerate(kfold.split(df, df.target)): \n    df.loc[valid_idx, 'fold'] = int(i)\n    \ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:35:06.628828Z","iopub.execute_input":"2022-10-28T05:35:06.629293Z","iopub.status.idle":"2022-10-28T05:35:06.651212Z","shell.execute_reply.started":"2022-10-28T05:35:06.629258Z","shell.execute_reply":"2022-10-28T05:35:06.649921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folds = df","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:35:06.654127Z","iopub.execute_input":"2022-10-28T05:35:06.654506Z","iopub.status.idle":"2022-10-28T05:35:06.660076Z","shell.execute_reply.started":"2022-10-28T05:35:06.654473Z","shell.execute_reply":"2022-10-28T05:35:06.658796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _bytes_feature(value):\n  \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n  if isinstance(value, type(tf.constant(0))):\n    value = value.numpy() # BytesList won't unpack a string from an EagerTensor.\n  return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))\n\ndef _float_feature(value):\n  \"\"\"Returns a float_list from a float / double.\"\"\"\n  return tf.train.Feature(float_list=tf.train.FloatList(value=[value]))\n\ndef _int64_feature(value):\n  \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n  return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:35:06.662083Z","iopub.execute_input":"2022-10-28T05:35:06.663193Z","iopub.status.idle":"2022-10-28T05:35:06.672179Z","shell.execute_reply.started":"2022-10-28T05:35:06.663145Z","shell.execute_reply":"2022-10-28T05:35:06.671102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def serialize_example_train(img, tgt, name):\n  feature = {\n      'spectrogram': _bytes_feature(img),\n      'target': _float_feature(tgt),\n      'id': _bytes_feature(name),\n  }\n  example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n  return example_proto.SerializeToString()\n\ndef serialize_example_test(img, name):\n  feature = {\n      'spectrogram': _bytes_feature(img),\n      'id': _bytes_feature(name),\n  }\n  example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n  return example_proto.SerializeToString()","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:35:06.673269Z","iopub.execute_input":"2022-10-28T05:35:06.673871Z","iopub.status.idle":"2022-10-28T05:35:06.688425Z","shell.execute_reply.started":"2022-10-28T05:35:06.673822Z","shell.execute_reply":"2022-10-28T05:35:06.687233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for f in range(FOLDS):\n    ct = (folds['fold'] == f).sum()\n    idx = folds[folds['fold'] == f].index\n\n    print(ct)\n    print('Writing TFRecord %i of %i...'%(f,ct))\n    \n    with tf.io.TFRecordWriter('train%.2i-%i.tfrec'%(f,ct)) as writer:\n        for _, i in enumerate((tqdm(idx))):\n            r = folds.iloc[i]\n            y = np.float32(r.target)\n            file_id = r.id\n\n            img = np.empty((384, 512, 2), dtype=np.float32)\n\n            filename = '%s/train/%s.hdf5' % (di, file_id)\n            with h5py.File(filename, 'r') as f:\n                g = f[file_id]\n\n                for ch, s in enumerate(['H1', 'L1']):\n                    a = g[s]['SFTs'][:] * 1e22  # Fourier coefficient complex64\n\n                    p = a.real**2 + a.imag**2  # power\n                    p /= np.mean(p)  # normalize\n                    p = cv2.resize(p, (4096, 384))\n                    p = np.mean(p.reshape(384, 512, 8), axis=2)  # compress 4096 -> 128\n\n                    img[...,ch] = p\n                    \n            serialized_img = tf.io.serialize_tensor(img)\n            example = serialize_example_train(serialized_img, y, str.encode(file_id))\n            writer.write(example)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:35:06.690195Z","iopub.execute_input":"2022-10-28T05:35:06.690715Z","iopub.status.idle":"2022-10-28T05:38:02.869522Z","shell.execute_reply.started":"2022-10-28T05:35:06.69067Z","shell.execute_reply":"2022-10-28T05:38:02.867828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(di + '/sample_submission.csv')\nct = len(test_df)\nidx = test_df.index\nf = 0\nprint(ct)\nprint('Writing TFRecord %i of %i...'%(f,ct))\n\nwith tf.io.TFRecordWriter('test%.2i-%i.tfrec'%(f,ct)) as writer:\n    for _, i in enumerate((tqdm(idx))):\n        r = test_df.iloc[i]\n        file_id = r.id\n\n        img = np.empty((384, 512, 2), dtype=np.float32)\n\n        filename = '%s/test/%s.hdf5' % (di, file_id)\n        with h5py.File(filename, 'r') as f:\n            g = f[file_id]\n\n            for ch, s in enumerate(['H1', 'L1']):\n                a = g[s]['SFTs'][:] * 1e22  # Fourier coefficient complex64\n\n                p = a.real**2 + a.imag**2  # power\n                p /= np.mean(p)  # normalize\n                p = cv2.resize(p, (4096, 384))\n                p = np.mean(p.reshape(384, 512, 8), axis=2)  # compress 4096 -> 128\n\n                img[...,ch] = p\n\n        serialized_img = tf.io.serialize_tensor(img)\n        example = serialize_example_test(serialized_img, str.encode(file_id))\n        writer.write(example)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:42:31.450057Z","iopub.execute_input":"2022-10-28T05:42:31.451511Z","iopub.status.idle":"2022-10-28T06:21:32.191911Z","shell.execute_reply.started":"2022-10-28T05:42:31.451441Z","shell.execute_reply":"2022-10-28T06:21:32.190019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}