{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-03T14:48:29.607042Z","iopub.execute_input":"2023-01-03T14:48:29.607719Z","iopub.status.idle":"2023-01-03T14:48:29.635019Z","shell.execute_reply.started":"2023-01-03T14:48:29.607567Z","shell.execute_reply":"2023-01-03T14:48:29.634012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U keras-efficientnet-v2","metadata":{"execution":{"iopub.status.busy":"2023-01-03T14:50:20.197992Z","iopub.execute_input":"2023-01-03T14:50:20.198699Z","iopub.status.idle":"2023-01-03T14:50:32.064701Z","shell.execute_reply.started":"2023-01-03T14:50:20.198656Z","shell.execute_reply":"2023-01-03T14:50:32.063254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras_efficientnet_v2","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import h5py\nfrom glob import glob\nfrom pathlib import Path","metadata":{"execution":{"iopub.status.busy":"2023-01-03T15:26:29.141349Z","iopub.execute_input":"2023-01-03T15:26:29.141748Z","iopub.status.idle":"2023-01-03T15:26:29.149161Z","shell.execute_reply.started":"2023-01-03T15:26:29.1417Z","shell.execute_reply":"2023-01-03T15:26:29.147822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib as mpl","metadata":{"execution":{"iopub.status.busy":"2023-01-03T15:26:30.873218Z","iopub.execute_input":"2023-01-03T15:26:30.873602Z","iopub.status.idle":"2023-01-03T15:26:30.879767Z","shell.execute_reply.started":"2023-01-03T15:26:30.873569Z","shell.execute_reply":"2023-01-03T15:26:30.878441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nfrom pathlib import Path\nimport shutil\nfrom tqdm.auto import tqdm\n","metadata":{"execution":{"iopub.status.busy":"2023-01-03T15:26:36.743089Z","iopub.execute_input":"2023-01-03T15:26:36.743492Z","iopub.status.idle":"2023-01-03T15:26:36.74905Z","shell.execute_reply.started":"2023-01-03T15:26:36.743457Z","shell.execute_reply":"2023-01-03T15:26:36.747797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"origin = '/kaggle/input/g2net-detecting-continuous-gravitational-waves/train/'","metadata":{"execution":{"iopub.status.busy":"2023-01-03T15:26:37.233041Z","iopub.execute_input":"2023-01-03T15:26:37.233583Z","iopub.status.idle":"2023-01-03T15:26:37.241034Z","shell.execute_reply.started":"2023-01-03T15:26:37.233532Z","shell.execute_reply":"2023-01-03T15:26:37.238879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count = 0\nfor root, folders, filenames in os.walk('/kaggle/input'):\n   print(root, folders)","metadata":{"execution":{"iopub.status.busy":"2023-01-03T15:26:37.893834Z","iopub.execute_input":"2023-01-03T15:26:37.894222Z","iopub.status.idle":"2023-01-03T15:26:42.884101Z","shell.execute_reply.started":"2023-01-03T15:26:37.894189Z","shell.execute_reply":"2023-01-03T15:26:42.882948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utils","metadata":{}},{"cell_type":"code","source":"# Utility to read hdf5 file\ndef read_data(file: Path):\n    with h5py.File(file, \"r\") as f:\n        file=Path(file)\n        filename = file.stem\n        f = f[filename]\n        h1 = f[\"H1\"]\n        l1 = f[\"L1\"]\n        #freq_hz = list(f[\"frequency_Hz\"])\n        ###\n        h1_stft = h1[\"SFTs\"][()]\n        ###\n        #h1_timestamp = h1[\"timestamps_GPS\"][()]\n        h1_timestamp =1\n        # H2 data\n        l1_stft = l1[\"SFTs\"][()]\n        #l1_timestamp = l1[\"timestamps_GPS\"][()]\n        l1_timestamp =2\n        return {\n            \"H1\": [h1_stft, h1_timestamp],\n            \"L1\": [l1_stft, l1_timestamp],\n            #\"freq_hz\": freq_hz\n        }\ndef read_data_c(file):\n    file = Path(file)\n    with h5py.File(file, \"r\") as f:\n        filename = file.stem\n        f = f[filename]\n        h1 = f[\"H1\"]\n        l1 = f[\"L1\"]\n        freq_hz = f[\"frequency_Hz\"]\n        \n        h1_stft = h1[\"SFTs\"][()]\n        h1_timestamp = h1[\"timestamps_GPS\"][()]\n        # H2 data\n        l1_stft = l1[\"SFTs\"][()]\n        l1_timestamp = l1[\"timestamps_GPS\"][()]\n        \n        return [h1_stft, h1_timestamp],            [l1_stft, l1_timestamp], np.array(freq_hz)\ndef power_spectrogram2(h1_sft,l1_sft):\n    i=0\n    \n    img = np.empty((360,360,3), dtype=np.float32)\n    try:\n        a=h1_sft[:360, :4320]*1e22\n        p = a.real**2 + a.imag**2\n        p /= np.mean(p)  # normalize\n        q=(a.real*2)*(np.angle(a))\n        q /= np.mean(q)  # normalize\n        try:\n            p = np.mean(p.reshape(360, 360,12), axis=2)\n            q = np.mean(q.reshape(360, 360,12), axis=2)\n        except:\n            a=h1_sft[:360, :3960]*1e22\n            p = a.real**2 + a.imag**2\n            q=(a.real*2)*(np.angle(a))\n            p /= np.mean(p)  # normalize\n            q /= np.mean(q)  # normalize\n            p = np.mean(p.reshape(360, 360,11), axis=2)\n            q = np.mean(q.reshape(360, 360,11), axis=2)\n\n        #normalized=(255*(p - np.min(p))/np.ptp(p)).astype(int)\n\n    #     normalized=(255*(p - np.min(p))/np.ptp(p)).astype(int)\n        img[:,:,0]= p*255\n#         img[:,:,2]= np.ones((360,360), dtype=np.float32)\n#         img[:,:,2]=img[:,:,2]*255\n\n\n        img[:,:,2]= q*255\n        a=l1_sft[:360, :4320]*1e22\n        p = a.real**2 + a.imag**2\n        p /= np.mean(p)  # normalize\n        try:\n            p = np.mean(p.reshape(360, 360,12), axis=2)\n        except:\n            a=l1_sft[:360, :3960]*1e22\n            p = a.real**2 + a.imag**2\n            p /= np.mean(p)  # normalize\n            p = np.mean(p.reshape(360, 360,11), axis=2)\n\n        #normalized=(255*(p - np.min(p))/np.ptp(p)).astype(int)\n        img[:,:,1]= p*255\n\n#         img = np.moveaxis(img, 0, -1)\n        #return np.asarray([img[0],img[1]]).astype('float32')\n    except:\n        print(\"empty image\")\n#         img = np.moveaxis(img, 0, -1)\n\n    return img\ndef show_power_spectrogram2(filename):\n    f = h5py.File(filename, 'r')\n\n  # read fourier transform coefficients\n    (h1_sfts, h1_ts), (l1_sfts, l1_ts), freq=read_data_c(filename)\n    img=power_spectrogram2(h1_sfts,l1_sfts)\n    plt.imshow(img)\n    plt.show()\ndef test_dataset(filename):\n    filename=Path(filename)\n    with h5py.File(filename) as f:\n                filename=filename.stem\n                g = f[filename]\n                print(filename)\n                try:\n                    for ch, s in enumerate(['H1', 'L1']):\n                            x=g[s]['SFTs'].shape\n                except:\n                    print('bad dataset')","metadata":{"execution":{"iopub.status.busy":"2023-01-03T15:26:42.886007Z","iopub.execute_input":"2023-01-03T15:26:42.886406Z","iopub.status.idle":"2023-01-03T15:26:42.917513Z","shell.execute_reply.started":"2023-01-03T15:26:42.886371Z","shell.execute_reply":"2023-01-03T15:26:42.915802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"tensor flow utils","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2023-01-03T15:26:42.918803Z","iopub.execute_input":"2023-01-03T15:26:42.919185Z","iopub.status.idle":"2023-01-03T15:26:42.933239Z","shell.execute_reply.started":"2023-01-03T15:26:42.919149Z","shell.execute_reply":"2023-01-03T15:26:42.932207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _bytes_feature(value):\n  \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n  if isinstance(value, type(tf.constant(0))):\n    value = value.numpy() # BytesList won't unpack a string from an EagerTensor.\n  return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))\ndef _float_feature(value):\n  \"\"\"Returns a float_list from a float / double.\"\"\"\n  return tf.train.Feature(float_list=tf.train.FloatList(value=[value]))\ndef _int64_feature(value):\n  \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n  return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))","metadata":{"execution":{"iopub.status.busy":"2023-01-03T15:26:42.935247Z","iopub.execute_input":"2023-01-03T15:26:42.937059Z","iopub.status.idle":"2023-01-03T15:26:42.946627Z","shell.execute_reply.started":"2023-01-03T15:26:42.937016Z","shell.execute_reply":"2023-01-03T15:26:42.945523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def serialize_example_train(img, tgt, name):\n  feature = {\n      'spectrogram': _bytes_feature(img),\n      'target': _float_feature(tgt),\n      'id': _bytes_feature(name),\n  }\n  example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n  return example_proto.SerializeToString()\n\ndef serialize_example_test(img, name):\n  feature = {\n      'spectrogram': _bytes_feature(img),\n      'id': _bytes_feature(name),\n  }\n  example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n  return example_proto.SerializeToString()","metadata":{"execution":{"iopub.status.busy":"2023-01-03T15:26:42.948202Z","iopub.execute_input":"2023-01-03T15:26:42.948854Z","iopub.status.idle":"2023-01-03T15:26:42.964215Z","shell.execute_reply.started":"2023-01-03T15:26:42.948816Z","shell.execute_reply":"2023-01-03T15:26:42.963075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Features","metadata":{}},{"cell_type":"code","source":"heigh=360\n# width=360\n# first_cut=4320 #360\n# compress_1=12\n# second_cut=3960 #360\n# compress_2=11\nwidth=128\nfirst_cut=4096\ncompress_1=32\nsecond_cut=4096-128\ncompress_2=31","metadata":{"execution":{"iopub.status.busy":"2023-01-03T15:26:43.850701Z","iopub.execute_input":"2023-01-03T15:26:43.851315Z","iopub.status.idle":"2023-01-03T15:26:43.856505Z","shell.execute_reply.started":"2023-01-03T15:26:43.85128Z","shell.execute_reply":"2023-01-03T15:26:43.855443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# load and prepare Train dataset ","metadata":{}},{"cell_type":"code","source":"# path_to_files=[] \n# names_of_files=[]\n# for root, folders, filenames in os.walk('/kaggle/input/g2net-detecting-continuous-gravitational-waves/test/'):\n#     path_to_files.append(root)\n#     names_of_files.append(filenames)\n       \n#print(root, folders,filenames)\n# test_df_complete=pd.DataFrame({'name_of_file':names_of_files[0]})\n# test_df_complete['radacina']='/kaggle/input/g2net-detecting-continuous-gravitational-waves/test/'\n# test_df_complete['cale_fisier']=test_df_complete['radacina']+test_df_complete['name_of_file']\n# test_df_complete['target']=0.5\n# test_df_complete['id']=test_df_complete['name_of_file'].str.replace('.hdf5', '')\n# test_df=pd.DataFrame()\n# test_df=test_df_complete[['id','target','cale_fisier']]\n# test_df=test_df.sort_values('id') ","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:48:35.381359Z","iopub.execute_input":"2022-12-24T16:48:35.382241Z","iopub.status.idle":"2022-12-24T16:48:35.388207Z","shell.execute_reply.started":"2022-12-24T16:48:35.382194Z","shell.execute_reply":"2022-12-24T16:48:35.387308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df=test_df.sample(100)\n# test_df=test_df.reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:48:35.862271Z","iopub.execute_input":"2022-12-24T16:48:35.862542Z","iopub.status.idle":"2022-12-24T16:48:35.868473Z","shell.execute_reply.started":"2022-12-24T16:48:35.862518Z","shell.execute_reply":"2022-12-24T16:48:35.867309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission['target'] = preds[:,0]\n# submission = submission.sort_values('id') \n# submission.to_csv('submission.csv', index=False)\n# submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:48:35.918834Z","iopub.execute_input":"2022-12-24T16:48:35.919835Z","iopub.status.idle":"2022-12-24T16:48:35.924265Z","shell.execute_reply.started":"2022-12-24T16:48:35.91979Z","shell.execute_reply":"2022-12-24T16:48:35.923274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.read_csv(\"/kaggle/input/g2net-detecting-continuous-gravitational-waves/train_labels.csv\")\nlen(train)","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:40:25.797516Z","iopub.execute_input":"2022-12-24T19:40:25.798303Z","iopub.status.idle":"2022-12-24T19:40:25.820894Z","shell.execute_reply.started":"2022-12-24T19:40:25.798265Z","shell.execute_reply":"2022-12-24T19:40:25.820104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# load and prepare Test dataset ","metadata":{}},{"cell_type":"code","source":"#test_df=pd.DataFrame()\ntest_df = pd.read_csv('../input/g2net-detecting-continuous-gravitational-waves/sample_submission.csv')\n#test_files = glob(f\"{ROOT_DIR}/test/*.hdf5\")\ntest_df['cale_fisier']='/kaggle/input/g2net-detecting-continuous-gravitational-waves/test/'+test_df['id']+'.hdf5'\n# test_df=test_df.sample(10)\n# test_df=test_df.reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:41:01.200111Z","iopub.execute_input":"2022-12-24T19:41:01.20053Z","iopub.status.idle":"2022-12-24T19:41:01.234591Z","shell.execute_reply.started":"2022-12-24T19:41:01.200495Z","shell.execute_reply":"2022-12-24T19:41:01.233371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def normalize_2d(matrix):\n    # Only this is changed to use 2-norm put 2 instead of 1\n    norm = np.linalg.norm(matrix, 1)\n    # normalized matrix\n    matrix = matrix/norm\n    \n    return matrix","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_WIDTH=128","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sft_to_img(sft, ts, buckets=IMG_WIDTH):\n    bucket_size = (ts.max() - ts.min()) // buckets\n    idx = np.searchsorted(ts, [ts[0] + bucket_size * i for i in range(buckets)])\n    # 1. ASD\n    sft = np.absolute(sft)    \n    # 2. shrink\n    global_noise_amp = np.mean(sft, axis=1)\n    img = np.stack([\n        np.mean(i, axis=1) if i.shape[1] > 0 else global_noise_amp for i in np.array_split(sft, idx[1:], axis=1) \n    ])    \n    ts_noise = img.mean(axis=1)\n    # 3. Normalize\n    mean, std = np.mean(img), np.std(img.astype(np.float64))\n    # print(mean, std)    \n\n    img = img - mean\n    img = img / std / 4\n    img *= 128 \n    img += 128\n    # img = img * (255 / np.max(img))\n#     img = np.clip(img, 0, 255).astype(np.uint8)\n    # print(np.min(img), np.max(img), np.mean(img), np.std(img))\n#     return img.T, ts_noise\n    return img.T\n    if DEBUG:\n        img, ts_noise = sft_to_img(h1_sft, h1_ts)\n        plt.figure()\n        plt.imshow(img, cmap='gray')\n        plt.figure()\n        plt.plot(ts_noise)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n#test_df = pd.read_csv(di + '/sample_submission.csv')\ntrain_only=False\nif train_only:\n    print('skip the nonstationary model dataset save')\nelse:\n    ct = len(test_df)\n    idx = test_df.index\n    f = 0\n    print(ct)\n    print('Writing TFRecord %i of %i...'%(f,ct))\n    xi=0\n    with tf.io.TFRecordWriter('test%.2i-%i.tfrec'%(f,ct)) as writer:\n        for _, i in enumerate((tqdm(idx))):\n            r = test_df.iloc[i]\n            file_id = r.id\n            filename=r.cale_fisier\n            img = np.empty((360, width, 2), dtype=np.float64)\n            (h1_sfts, h1_ts), (l1_sfts, l1_ts), freq=read_data_c(filename)\n#             DEBUG=True\n            #sft_to_img(h1_sfts, h1_ts, buckets=IMG_WIDTH)\n            imagine=sft_to_img(h1_sfts, h1_ts, buckets=IMG_WIDTH)\n            img[...,0] = imagine\n            imagine=sft_to_img(l1_sfts, l1_ts, buckets=IMG_WIDTH)\n            img[...,1] = imagine\n#           for _, i in enumerate((tqdm(idx))):\n#             r = test_df.iloc[i]\n#             file_id = r.id\n#             filename=r.cale_fisier\n#             img = np.empty((360, width, 2), dtype=np.float64)\n\n#             #filename = '%s/test/%s.hdf5' % (di, file_id)\n#             with h5py.File(filename, 'r') as f:\n#                 g = f[file_id]\n#                 #print('unu')\n#                 for ch, s in enumerate(['H1', 'L1']):\n#                     print(xi)\n#                     xi=xi+1\n#                     try:\n#                         a = np.array(g[s]['SFTs'], dtype=np.complex128)\n#                         a = a[:360, :first_cut]   # Fourier coefficient complex64\n#                         p = 2*(a.real**2 + a.imag**2)  # power\n#                         p = normalize_2d(p)\n#                         p = np.mean(p.reshape(360, width,compress_1), axis=2)\n#                     except:\n#                         a = np.array(g[s]['SFTs'], dtype=np.complex128)\n#                         a = a[:360, :second_cut]  # Fourier coefficient complex64\n#                         p = 2*(a.real**2 + a.imag**2)  # power\n#                         p = normalize_2d(p)\n#                         p = np.mean(p.reshape(360,width,compress_2), axis=2)\n#                     img[...,ch] = p\n\n\n    #             try:\n    #                 a=g[\"H1\"]\n    #                 a=a['SFTs'][:360, :4320]*1e22\n    #                 #a = g['H1']['SFTs'][:360, :4320]# Fourier coefficient complex64\n    #                 q = np.sin(np.angle(a))*(np.abs(a)**2)  # power\n    #                 q /= np.mean(q)  # normalize\n    #                 q = np.mean(q.reshape(360, 360,12), axis=2)\n    #                 print(np.min(q),np.max(q))\n    #                 scaler = MinMaxScaler(feature_range=(0, 255))\n    #                 scaler = scaler.fit(q)\n    #                 img[...,2] = q\n    #                 #cv2.imwrite(\"blabla.jpg\",img)\n    #             except:\n    #                 a=g[\"H1\"]\n    #                 a=a['SFTs'][:360, :3960]*1e22\n    #                 p = np.sin(np.angle(a))*(np.abs(a)**2)  # power\n    #                 p /= np.mean(p)  # normalize\n    #                 p = np.mean(p.reshape(360, 360,11), axis=2)\n    #                 scaler = scaler.fit(p)\n    #                 img[...,2] = p  \n            serialized_img = tf.io.serialize_tensor(img)\n            example = serialize_example_test(serialized_img, str.encode(file_id))\n            writer.write(example)\ntrain_only=False","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:41:05.453066Z","iopub.execute_input":"2022-12-24T19:41:05.453794Z","iopub.status.idle":"2022-12-24T19:41:09.136493Z","shell.execute_reply.started":"2022-12-24T19:41:05.453747Z","shell.execute_reply":"2022-12-24T19:41:09.134829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:49:08.915753Z","iopub.execute_input":"2022-12-24T16:49:08.916157Z","iopub.status.idle":"2022-12-24T16:49:08.920771Z","shell.execute_reply.started":"2022-12-24T16:49:08.916118Z","shell.execute_reply":"2022-12-24T16:49:08.919643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# path_to_files=list()  \n# names_of_files=list()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:49:08.922427Z","iopub.execute_input":"2022-12-24T16:49:08.922823Z","iopub.status.idle":"2022-12-24T16:49:08.936863Z","shell.execute_reply.started":"2022-12-24T16:49:08.922783Z","shell.execute_reply":"2022-12-24T16:49:08.935849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# path_to_files=[] \n# names_of_files=[]","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:49:08.939344Z","iopub.execute_input":"2022-12-24T16:49:08.939705Z","iopub.status.idle":"2022-12-24T16:49:08.945487Z","shell.execute_reply.started":"2022-12-24T16:49:08.939668Z","shell.execute_reply":"2022-12-24T16:49:08.944724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# for root, folders, filenames in os.walk('/kaggle/input/g2net-detecting-continuous-gravitational-waves/test/'):\n#     path_to_files.append(root)\n#     names_of_files.append(filenames)\n       \n# #print(root, folders,filenames)","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:49:08.946565Z","iopub.execute_input":"2022-12-24T16:49:08.946919Z","iopub.status.idle":"2022-12-24T16:49:08.954432Z","shell.execute_reply.started":"2022-12-24T16:49:08.94689Z","shell.execute_reply":"2022-12-24T16:49:08.953699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df_complete=pd.DataFrame({'name_of_file':names_of_files[0]})","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:49:08.957559Z","iopub.execute_input":"2022-12-24T16:49:08.958259Z","iopub.status.idle":"2022-12-24T16:49:08.967016Z","shell.execute_reply.started":"2022-12-24T16:49:08.958231Z","shell.execute_reply":"2022-12-24T16:49:08.966062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df_complete['radacina']='/kaggle/input/g2net-detecting-continuous-gravitational-waves/test/'","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:49:08.968188Z","iopub.execute_input":"2022-12-24T16:49:08.968615Z","iopub.status.idle":"2022-12-24T16:49:08.975931Z","shell.execute_reply.started":"2022-12-24T16:49:08.968588Z","shell.execute_reply":"2022-12-24T16:49:08.974811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df_complete['cale_fisier']=test_df_complete['radacina']+test_df_complete['name_of_file']\n# test_df_complete['target']=0.5","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:49:08.977146Z","iopub.execute_input":"2022-12-24T16:49:08.977457Z","iopub.status.idle":"2022-12-24T16:49:08.985446Z","shell.execute_reply.started":"2022-12-24T16:49:08.977432Z","shell.execute_reply":"2022-12-24T16:49:08.984335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df_complete['id']=test_df_complete['name_of_file'].str.replace('.hdf5', '')","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:49:08.986554Z","iopub.execute_input":"2022-12-24T16:49:08.986833Z","iopub.status.idle":"2022-12-24T16:49:08.994441Z","shell.execute_reply.started":"2022-12-24T16:49:08.986808Z","shell.execute_reply":"2022-12-24T16:49:08.99367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df_complete.tail()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:49:08.999033Z","iopub.execute_input":"2022-12-24T16:49:08.999482Z","iopub.status.idle":"2022-12-24T16:49:09.003722Z","shell.execute_reply.started":"2022-12-24T16:49:08.999454Z","shell.execute_reply":"2022-12-24T16:49:09.002728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cols = list(test_df_complete.columns.values)","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:49:09.005246Z","iopub.execute_input":"2022-12-24T16:49:09.005933Z","iopub.status.idle":"2022-12-24T16:49:09.012696Z","shell.execute_reply.started":"2022-12-24T16:49:09.005895Z","shell.execute_reply":"2022-12-24T16:49:09.011752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df=test_df_complete[['id','target','cale_fisier']]","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:49:09.013901Z","iopub.execute_input":"2022-12-24T16:49:09.014645Z","iopub.status.idle":"2022-12-24T16:49:09.020938Z","shell.execute_reply.started":"2022-12-24T16:49:09.014606Z","shell.execute_reply":"2022-12-24T16:49:09.019843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df=test_df.sample(400)\n# test_df=test_df.reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:49:09.022014Z","iopub.execute_input":"2022-12-24T16:49:09.022478Z","iopub.status.idle":"2022-12-24T16:49:09.028813Z","shell.execute_reply.started":"2022-12-24T16:49:09.022441Z","shell.execute_reply":"2022-12-24T16:49:09.027701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:41:18.69238Z","iopub.execute_input":"2022-12-24T19:41:18.692764Z","iopub.status.idle":"2022-12-24T19:41:18.710247Z","shell.execute_reply.started":"2022-12-24T19:41:18.692732Z","shell.execute_reply":"2022-12-24T19:41:18.709106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import StratifiedKFold","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:49:09.052936Z","iopub.execute_input":"2022-12-24T16:49:09.053289Z","iopub.status.idle":"2022-12-24T16:49:09.059177Z","shell.execute_reply.started":"2022-12-24T16:49:09.053263Z","shell.execute_reply":"2022-12-24T16:49:09.058158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FOLDS=5\nSEED=47","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:41:35.662159Z","iopub.execute_input":"2022-12-24T19:41:35.662554Z","iopub.status.idle":"2022-12-24T19:41:35.667631Z","shell.execute_reply.started":"2022-12-24T19:41:35.662522Z","shell.execute_reply":"2022-12-24T19:41:35.666415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Load a model","metadata":{}},{"cell_type":"code","source":"from keras_preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import layers, Model, Input, losses, metrics, optimizers, callbacks","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:41:43.968677Z","iopub.execute_input":"2022-12-24T19:41:43.969339Z","iopub.status.idle":"2022-12-24T19:41:44.962994Z","shell.execute_reply.started":"2022-12-24T19:41:43.969285Z","shell.execute_reply":"2022-12-24T19:41:44.961882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_labeled_tfrecord(example):\n    tfrec_format = {\n        'spectrogram'          : tf.io.FixedLenFeature([], tf.string),\n        'target'               : tf.io.FixedLenFeature([], tf.float32),\n        'id'                   : tf.io.FixedLenFeature([], tf.string),\n    }           \n    example = tf.io.parse_single_example(example, tfrec_format)\n    example['spectrogram'] = tf.io.parse_tensor(example['spectrogram'],out_type=tf.float64)\n    example['spectrogram'] = tf.reshape(example['spectrogram'], [*IMG_SIZE,2])\n    #print(example['spectrogram'].shape)\n    return example['spectrogram'], example['target']\n\n\ndef read_unlabeled_tfrecord(example, return_image_name):\n    tfrec_format = {\n        'spectrogram'          : tf.io.FixedLenFeature([], tf.string),\n        'id'                   : tf.io.FixedLenFeature([], tf.string)\n    }\n    example = tf.io.parse_single_example(example, tfrec_format)\n    example['spectrogram'] = tf.io.parse_tensor(example['spectrogram'],out_type=tf.float64)\n    example['spectrogram'] = tf.reshape(example['spectrogram'], [*IMG_SIZE,2])\n    return example['spectrogram'], example['id'] if return_image_name else 0\n\n \ndef prepare_image(img, augment=False):    \n    if augment:\n        img = augmentation(img)\n# for testing withou a dummy function should be created: \n                                            # def augmentation(img):\n                                            #     img = img\n                                            #     return img\n    #The tf.reshape does not change the order of or the total number of elements in the tensor, \n    #and so it can reuse the underlying data buffer. \n    #This makes it a fast operation independent of how big of a tensor it is operating on.\n    img = tf.reshape(img, [*IMG_SIZE, 2])\n            \n    return img\n\ndef count_data_items(filenames):\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) \n         for filename in filenames]\n    return np.sum(n)\ndef get_dataset(files, augment = False, shuffle = False, repeat = False, \n                labeled=True, return_image_names=True, batch_size=32):\n    \n    ds = tf.data.TFRecordDataset(files, num_parallel_reads=AUTO)\n    ds = ds.cache()\n    \n    if repeat:\n        ds = ds.repeat()\n    \n    if shuffle: \n        ds = ds.shuffle(1024*8 if tpu else 128)\n        opt = tf.data.Options()\n        opt.experimental_deterministic = False\n        ds = ds.with_options(opt)\n        \n    if labeled: \n        ds = ds.map(read_labeled_tfrecord, num_parallel_calls=AUTO)\n    else:\n        ds = ds.map(lambda example: read_unlabeled_tfrecord(example, return_image_names), \n                    num_parallel_calls=AUTO)      \n    #Augumentation here\n    ds = ds.map(lambda img, imgname_or_label: (prepare_image(img, augment=augment,), \n                                               imgname_or_label), \n                num_parallel_calls=AUTO)\n    \n    ds = ds.batch(batch_size)\n    ds = ds.prefetch(AUTO)\n    return ds","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:41:50.263736Z","iopub.execute_input":"2022-12-24T19:41:50.264561Z","iopub.status.idle":"2022-12-24T19:41:50.287245Z","shell.execute_reply.started":"2022-12-24T19:41:50.264517Z","shell.execute_reply":"2022-12-24T19:41:50.285961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q efficientnet >> /dev/null","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:41:50.971919Z","iopub.execute_input":"2022-12-24T19:41:50.972605Z","iopub.status.idle":"2022-12-24T19:42:04.372589Z","shell.execute_reply.started":"2022-12-24T19:41:50.97256Z","shell.execute_reply":"2022-12-24T19:42:04.371386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEVICE = \"TPU\" #or \"GPU\"\n\n# USE DIFFERENT SEED FOR DIFFERENT STRATIFIED KFOLD\nSEED = 42\n\nFOLDS = 5\nIMG_SIZE = [360,128]\n\nBATCH_SIZE = 8\nEPOCH = 45\n\n# TEST TIME AUGMENTATION STEPS\nTTA = 1","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:42:04.375487Z","iopub.execute_input":"2022-12-24T19:42:04.375944Z","iopub.status.idle":"2022-12-24T19:42:04.383496Z","shell.execute_reply.started":"2022-12-24T19:42:04.375899Z","shell.execute_reply":"2022-12-24T19:42:04.382109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to get hardware strategy\ndef get_hardware_strategy():\n    try:\n        # TPU detection. No parameters necessary if TPU_NAME environment variable is\n        # set: this is always the case on Kaggle.\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        print('Running on TPU ', tpu.master())\n    except ValueError:\n        tpu = None\n\n    if tpu:\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        #policy = mixed_precision.Policy('mixed_bfloat16')\n        #mixed_precision.set_global_policy(policy)\n        tf.config.optimizer.set_jit(True)\n    else:\n        # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n        strategy = tf.distribute.get_strategy()\n\n    print(\"REPLICAS: \", strategy.num_replicas_in_sync)\n    return tpu, strategy\n\ntpu, strategy = get_hardware_strategy()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:42:04.384976Z","iopub.execute_input":"2022-12-24T19:42:04.385456Z","iopub.status.idle":"2022-12-24T19:42:04.402Z","shell.execute_reply.started":"2022-12-24T19:42:04.385419Z","shell.execute_reply":"2022-12-24T19:42:04.400584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\nBATCH_SIZE *= strategy.num_replicas_in_sync","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:42:24.716929Z","iopub.execute_input":"2022-12-24T19:42:24.717411Z","iopub.status.idle":"2022-12-24T19:42:24.722746Z","shell.execute_reply.started":"2022-12-24T19:42:24.717374Z","shell.execute_reply":"2022-12-24T19:42:24.721423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf, re, math\nimport tensorflow.keras.backend as K\n# import efficientnet.tfkeras as efn\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import roc_auc_score\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:42:29.335724Z","iopub.execute_input":"2022-12-24T19:42:29.336192Z","iopub.status.idle":"2022-12-24T19:42:30.140491Z","shell.execute_reply.started":"2022-12-24T19:42:29.336152Z","shell.execute_reply":"2022-12-24T19:42:30.139248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GCS_PATH_STRATIFICATED='/kaggle/working'","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:42:36.416692Z","iopub.execute_input":"2022-12-24T19:42:36.417115Z","iopub.status.idle":"2022-12-24T19:42:36.423044Z","shell.execute_reply.started":"2022-12-24T19:42:36.417083Z","shell.execute_reply":"2022-12-24T19:42:36.421652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files_test  = tf.io.gfile.glob(GCS_PATH_STRATIFICATED + '/test*.tfrec')","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:42:43.55338Z","iopub.execute_input":"2022-12-24T19:42:43.55382Z","iopub.status.idle":"2022-12-24T19:42:43.560908Z","shell.execute_reply.started":"2022-12-24T19:42:43.553784Z","shell.execute_reply":"2022-12-24T19:42:43.559716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test samples images","metadata":{}},{"cell_type":"code","source":"# FNAME = tf.io.gfile.glob([GCS_PATH_STRATIFICATED_modified + '/train*.tfrec'])\nrow = 6; col = 2;\nrow = min(row,6//col)\n\n# all_elements = get_dataset(FNAME, augment=False, batch_size=32).unbatch()\n# augmented_element = all_elements.repeat().batch(32)\nall_elements = get_dataset(files_test,labeled=False,return_image_names=False,augment=False,shuffle=False, batch_size=32)\n\n          \n# for (img,mask) in augmented_element:\nfor (img,mask) in all_elements:\n    print(img.shape)\n    img = (img-np.min(img))/(np.max(img)-np.min(img)+1e-5)\n    plt.figure(figsize=(15,int(15*row/col)))\n\n    i=0\n    j=1\n    while j<=row*col:\n        plt.subplot(row,col,j)\n        plt.axis('on')\n\n        plt.imshow(img[i,:,:,0])\n        j += 1\n        \n        plt.subplot(row,col,j)\n        plt.axis('on')\n        plt.imshow(img[i,:,:,1])\n        j += 1\n\n        i += 1\n\n    plt.show()\n    break","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:42:57.87753Z","iopub.execute_input":"2022-12-24T19:42:57.877984Z","iopub.status.idle":"2022-12-24T19:42:59.256472Z","shell.execute_reply.started":"2022-12-24T19:42:57.877948Z","shell.execute_reply":"2022-12-24T19:42:59.255355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build or compile the model  functions","metadata":{}},{"cell_type":"code","source":"# def build_model():\n#     inp = tf.keras.layers.Input(shape=(*IMG_SIZE, 2))\n#     in_conv = tf.keras.layers.Conv2D(3, 3, strides=(1, 1), padding='same')\n#     base = efn.EfficientNetB7(input_shape=(*IMG_SIZE, 3), weights='imagenet', include_top=False)\n# #     base = tf.keras.applications.EfficientNetB7(input_shape=(*IMG_SIZE, 3), weights='imagenet', include_top=False)\n#     x = in_conv(inp)\n#     x = base(x)\n#     x = tf.keras.layers.GlobalAveragePooling2D()(x)\n#     x = tf.keras.layers.Dropout(.2)(x)\n#     x = tf.keras.layers.Dense(1,activation='sigmoid')(x)\n#     model = tf.keras.Model(inputs=inp, outputs=x)\n#     opt = tf.keras.optimizers.Adam(learning_rate=0.001)\n# #     opt = tf.keras.optimizers.SGD(learning_rate=0.001,clipvalue=1)\n# #     opt = tf.keras.optimizers.RMSprop(learning_rate=0.001,clipvalue=1.)\n#     loss = tf.keras.losses.BinaryCrossentropy(label_smoothing=1e-5) \n#     model.compile(optimizer=opt, loss=loss, metrics=['AUC'])\n#     return model\n\n\ndef build_model():\n    inp = tf.keras.layers.Input(shape=(*IMG_SIZE, 2))\n    in_conv = tf.keras.layers.Conv2D(3, 7, strides=(1, 1), padding='same')\n#     base = keras_efficientnet_v2.EfficientNetV2S(pretrained=\"imagenet\")\n    base = keras_efficientnet_v2.EfficientNetV2S(input_shape=(*IMG_SIZE, 3), drop_connect_rate=0.2, num_classes=0, pretrained=\"imagenet21k-ft1k\")\n#     base = efn.EfficientNetB7(input_shape=(*IMG_SIZE, 3), weights='imagenet', include_top=False)\n    x = in_conv(inp)\n    x = base(x)\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    x = tf.keras.layers.Dropout(.3)(x)\n    x = tf.keras.layers.Dense(1,activation='sigmoid')(x)\n    model = tf.keras.Model(inputs=inp, outputs=x)\n    opt = tf.keras.optimizers.Adam(learning_rate=0.001)\n#     opt = tf.keras.optimizers.SGD(learning_rate=0.001,clipvalue=1)\n#     opt = tf.keras.optimizers.RMSprop(learning_rate=0.001,clipvalue=1.)\n    loss = tf.keras.losses.BinaryCrossentropy(label_smoothing=1e-5) \n    model.compile(optimizer=opt, loss=loss, metrics=['AUC'])\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:43:04.781105Z","iopub.execute_input":"2022-12-24T19:43:04.782152Z","iopub.status.idle":"2022-12-24T19:43:04.791229Z","shell.execute_reply.started":"2022-12-24T19:43:04.782109Z","shell.execute_reply":"2022-12-24T19:43:04.789861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## custom call back","metadata":{}},{"cell_type":"code","source":"def get_lr_callback():\n    lr_start   = 5e-5\n    lr_max     = 5e-4\n    lr_min     = 1e-5\n    lr_ramp_ep = 4\n    lr_sus_ep  = 4\n    lr_decay   = 0.9\n   \n    def lrfn(epoch):\n        if epoch < lr_ramp_ep:\n            lr = (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start     \n        elif epoch < lr_ramp_ep + lr_sus_ep:\n            lr = lr_max\n        else:\n            lr = (lr_max - lr_min) * lr_decay**(epoch - lr_ramp_ep - lr_sus_ep) + lr_min  \n        return lr\n    lr_callback = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=False)\n    return lr_callback","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:43:07.734948Z","iopub.execute_input":"2022-12-24T19:43:07.735385Z","iopub.status.idle":"2022-12-24T19:43:07.743899Z","shell.execute_reply.started":"2022-12-24T19:43:07.73535Z","shell.execute_reply":"2022-12-24T19:43:07.742662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:43:10.359737Z","iopub.execute_input":"2022-12-24T19:43:10.360246Z","iopub.status.idle":"2022-12-24T19:43:10.366431Z","shell.execute_reply.started":"2022-12-24T19:43:10.360204Z","shell.execute_reply":"2022-12-24T19:43:10.364798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect(generation=2) \ngc.collect(generation=1) \ngc.collect(generation=0) ","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:43:13.843539Z","iopub.execute_input":"2022-12-24T19:43:13.843983Z","iopub.status.idle":"2022-12-24T19:43:14.059849Z","shell.execute_reply.started":"2022-12-24T19:43:13.843947Z","shell.execute_reply":"2022-12-24T19:43:14.058491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sample_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:49:21.223249Z","iopub.execute_input":"2022-12-24T16:49:21.224442Z","iopub.status.idle":"2022-12-24T16:49:21.229221Z","shell.execute_reply.started":"2022-12-24T16:49:21.224406Z","shell.execute_reply":"2022-12-24T16:49:21.228096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var=0\nTTA=1\n# files_test = tf.io.gfile.glob(GCS_PATH_STRATIFICATED + '/test*.tfrec')\n# ds_test = get_dataset(files_test,labeled=False,return_image_names=False,augment=False,\n#             repeat=True,shuffle=False,batch_size=BATCH_SIZE*2)\n# ct_test = count_data_items(files_test); STEPS = TTA * ct_test/BATCH_SIZE/2","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:43:18.961163Z","iopub.execute_input":"2022-12-24T19:43:18.961567Z","iopub.status.idle":"2022-12-24T19:43:18.966491Z","shell.execute_reply.started":"2022-12-24T19:43:18.961534Z","shell.execute_reply":"2022-12-24T19:43:18.965505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_test = get_dataset(files_test,labeled=False,return_image_names=False,augment=False,\n        repeat=True,shuffle=False,batch_size=BATCH_SIZE*4)\nct_test = count_data_items(files_test); STEPS = TTA * ct_test/BATCH_SIZE/4","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:43:20.000518Z","iopub.execute_input":"2022-12-24T19:43:20.000909Z","iopub.status.idle":"2022-12-24T19:43:20.035201Z","shell.execute_reply.started":"2022-12-24T19:43:20.000878Z","shell.execute_reply":"2022-12-24T19:43:20.034073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ct_test","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:43:23.812674Z","iopub.execute_input":"2022-12-24T19:43:23.813074Z","iopub.status.idle":"2022-12-24T19:43:23.820901Z","shell.execute_reply.started":"2022-12-24T19:43:23.813042Z","shell.execute_reply":"2022-12-24T19:43:23.819582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  K.clear_session()\n# with strategy.scope():\n#     model = build_model()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:49:21.279768Z","iopub.execute_input":"2022-12-24T16:49:21.28041Z","iopub.status.idle":"2022-12-24T16:49:21.286426Z","shell.execute_reply.started":"2022-12-24T16:49:21.280371Z","shell.execute_reply":"2022-12-24T16:49:21.285316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.load_weights(f'/kaggle/input/low-g2net-tf-keras-train-test/fold-3.h5')","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:49:21.287876Z","iopub.execute_input":"2022-12-24T16:49:21.288558Z","iopub.status.idle":"2022-12-24T16:49:21.294185Z","shell.execute_reply.started":"2022-12-24T16:49:21.288518Z","shell.execute_reply":"2022-12-24T16:49:21.293361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission=test_df","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:43:29.314761Z","iopub.execute_input":"2022-12-24T19:43:29.315177Z","iopub.status.idle":"2022-12-24T19:43:29.320576Z","shell.execute_reply.started":"2022-12-24T19:43:29.315138Z","shell.execute_reply":"2022-12-24T19:43:29.319323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files_test","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:43:36.779193Z","iopub.execute_input":"2022-12-24T19:43:36.779587Z","iopub.status.idle":"2022-12-24T19:43:36.786796Z","shell.execute_reply.started":"2022-12-24T19:43:36.779555Z","shell.execute_reply":"2022-12-24T19:43:36.785576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# USE VERBOSE=0 for silent, VERBOSE=1 for interactive, VERBOSE=2 for commit\nVERBOSE = 2 if tpu else 1\nDISPLAY_PLOT = True\nvar=0\nTTA=1\noof_pred = []; oof_tar = []; oof_val = []; oof_names = []; oof_folds = [] \npreds = np.zeros((count_data_items(files_test),1))\nfor i in range(5):\n    K.clear_session()\n    with strategy.scope():\n        model = build_model()\n    model.load_weights(f'/kaggle/input/eff2nonstataugumented-low-g2net-tf-keras-train/fold-{var}.h5'.format(var))\n\n#     ds_test = get_dataset(files_test,labeled=False,return_image_names=False,augment=False,\n#                 repeat=True,shuffle=False,batch_size=BATCH_SIZE*2)\n#     ct_test = count_data_items(files_test); STEPS = TTA * ct_test/BATCH_SIZE/2\n#     ds_test = get_dataset(files_test,labeled=False,return_image_names=False,augment=False,\n#             repeat=True,shuffle=False,batch_size=BATCH_SIZE)\n#     ct_test = count_data_items(files_test); STEPS = TTA * ct_test/BATCH_SIZE\n    pred = model.predict(ds_test,steps=STEPS,verbose=VERBOSE)[:TTA*ct_test,]\n  \n    preds[:,0] += np.mean(pred.reshape((ct_test,TTA),order='F'),axis=1) / FOLDS\n    sample_submission[[f'preds{var}'.format(var)]]=pred\n    print('[=============Fold=================]')\n    print(f'/kaggle/input/low-g2net-tf-keras-train-test/fold-{var}.h5'.format(var))\n    var=var+1\n    print(f'preds{var}'.format(var))\n    K.clear_session()\n    gc.collect()\n    try:\n        del model\n    except:\n        print('no model here')\n#     try:\n#         del ds_test\n#     except:\n#         print('no ds here')\n#     try:\n#         del ct_test\n#     except:\n#         print('no ct here')\n#     try:\n#         del files_test\n#     except:\n#         print('no files_test')\n    gc.collect\n    print('gc collect done')","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:43:41.026955Z","iopub.execute_input":"2022-12-24T19:43:41.027339Z","iopub.status.idle":"2022-12-24T19:45:53.148284Z","shell.execute_reply.started":"2022-12-24T19:43:41.027306Z","shell.execute_reply":"2022-12-24T19:45:53.147006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### To decide if the file is needed","metadata":{}},{"cell_type":"code","source":"# submission = pd.read_csv('../input/g2net-detecting-continuous-gravitational-waves/sample_submission.csv')\n# submission.tail()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:58:56.48301Z","iopub.execute_input":"2022-12-24T16:58:56.48396Z","iopub.status.idle":"2022-12-24T16:58:56.487799Z","shell.execute_reply.started":"2022-12-24T16:58:56.483922Z","shell.execute_reply":"2022-12-24T16:58:56.487114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/g2net-detecting-continuous-gravitational-waves/sample_submission.csv')\n# submission.head()\n# submission=test_df[['id','target']]\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:46:05.040194Z","iopub.execute_input":"2022-12-24T19:46:05.040654Z","iopub.status.idle":"2022-12-24T19:46:05.055484Z","shell.execute_reply.started":"2022-12-24T19:46:05.040616Z","shell.execute_reply":"2022-12-24T19:46:05.054343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Compare head and tail\n\n1. \tid\ttarget\n*  0\t00054c878\t0.5\n*  1\t0007285a3\t0.5\n*  2\t00076c5a6\t0.5\n*  3\t001349290\t0.5\n*  4\t001a52e92\t0.5\n\n\n* id\ttarget\n* 7970\tffbce04ef\t0.5\n* 7971\tffc2d976b\t0.5\n* 7972\tffc905909\t0.5\n* 7973\tffe276f3e\t0.5\n* 7974\tfffa17f67\t0.5\n","metadata":{}},{"cell_type":"code","source":"submission.tail()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:45:53.173331Z","iopub.execute_input":"2022-12-24T19:45:53.173799Z","iopub.status.idle":"2022-12-24T19:45:53.191268Z","shell.execute_reply.started":"2022-12-24T19:45:53.173762Z","shell.execute_reply":"2022-12-24T19:45:53.189999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['target'] = preds[:,0]\nsubmission = submission.sort_values('id') \nsubmission.to_csv('submission.csv', index=False)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:46:09.801764Z","iopub.execute_input":"2022-12-24T19:46:09.80222Z","iopub.status.idle":"2022-12-24T19:46:09.819616Z","shell.execute_reply.started":"2022-12-24T19:46:09.802172Z","shell.execute_reply":"2022-12-24T19:46:09.818474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:46:14.119358Z","iopub.execute_input":"2022-12-24T19:46:14.119777Z","iopub.status.idle":"2022-12-24T19:46:14.19324Z","shell.execute_reply.started":"2022-12-24T19:46:14.119743Z","shell.execute_reply":"2022-12-24T19:46:14.192274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(submission.target,bins=100)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T19:46:17.583078Z","iopub.execute_input":"2022-12-24T19:46:17.583505Z","iopub.status.idle":"2022-12-24T19:46:17.95745Z","shell.execute_reply.started":"2022-12-24T19:46:17.583467Z","shell.execute_reply":"2022-12-24T19:46:17.956294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # USE VERBOSE=0 for silent, VERBOSE=1 for interactive, VERBOSE=2 for commit\n# VERBOSE = 2 if tpu else 1\n# DISPLAY_PLOT = True\n# var=2\n# pred = model.predict(ds_test,steps=STEPS,verbose=VERBOSE)[:TTA*ct_test,]\n# sample_submission[[f'preds{var}'.format(var)]]=pred\n# print(f'/kaggle/input/g2net-tf-keras-train-test/fold-{var}.h5'.format(var))\n# var=var+1\n# print(f'preds{var}'.format(var))\n# K.clear_session()\n# gc.collect()\n# try:\n#     del model\n# except:\n#     print('no model here')","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:50:21.532672Z","iopub.status.idle":"2022-12-24T16:50:21.533033Z","shell.execute_reply.started":"2022-12-24T16:50:21.532871Z","shell.execute_reply":"2022-12-24T16:50:21.532888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# try:\n#     del model\n# except:\n#     print('no model here')\n# try:\n#     del ds_test\n# except:\n#     print('no ds here')\n# try:\n#     del ct_test\n# except:\n#     print('no ct here')\n# try:\n#     del files_test\n# except:\n#     print('no files_test')\n# gc.collect","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:50:21.534186Z","iopub.status.idle":"2022-12-24T16:50:21.534606Z","shell.execute_reply.started":"2022-12-24T16:50:21.5344Z","shell.execute_reply":"2022-12-24T16:50:21.534418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# targets=(sample_submission['preds0']+sample_submission['preds1']+sample_submission['preds2']+sample_submission['preds3']+sample_submission['preds4'])/4","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:50:21.53579Z","iopub.status.idle":"2022-12-24T16:50:21.536342Z","shell.execute_reply.started":"2022-12-24T16:50:21.536162Z","shell.execute_reply":"2022-12-24T16:50:21.53618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sns.distplot(targets)","metadata":{"execution":{"iopub.status.busy":"2022-12-24T16:46:18.005268Z","iopub.status.idle":"2022-12-24T16:46:18.005815Z","shell.execute_reply.started":"2022-12-24T16:46:18.005513Z","shell.execute_reply":"2022-12-24T16:46:18.005539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df.target=targets","metadata":{"execution":{"iopub.status.busy":"2022-12-24T12:31:03.264005Z","iopub.execute_input":"2022-12-24T12:31:03.264632Z","iopub.status.idle":"2022-12-24T12:31:03.273083Z","shell.execute_reply.started":"2022-12-24T12:31:03.264582Z","shell.execute_reply":"2022-12-24T12:31:03.270973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission=pd.DataFrame()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T12:31:06.187856Z","iopub.execute_input":"2022-12-24T12:31:06.189087Z","iopub.status.idle":"2022-12-24T12:31:06.195433Z","shell.execute_reply.started":"2022-12-24T12:31:06.189031Z","shell.execute_reply":"2022-12-24T12:31:06.194252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission[['id','target']]=test_df[['id','target']]","metadata":{"execution":{"iopub.status.busy":"2022-12-24T11:59:21.698864Z","iopub.status.idle":"2022-12-24T11:59:21.699276Z","shell.execute_reply.started":"2022-12-24T11:59:21.699067Z","shell.execute_reply":"2022-12-24T11:59:21.699088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submission[['id','target']]=sample_submission[['id','target']]","metadata":{"execution":{"iopub.status.busy":"2022-12-24T11:59:21.700819Z","iopub.status.idle":"2022-12-24T11:59:21.701423Z","shell.execute_reply.started":"2022-12-24T11:59:21.701104Z","shell.execute_reply":"2022-12-24T11:59:21.701139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-24T12:31:10.611003Z","iopub.execute_input":"2022-12-24T12:31:10.612Z","iopub.status.idle":"2022-12-24T12:31:10.62121Z","shell.execute_reply.started":"2022-12-24T12:31:10.611949Z","shell.execute_reply":"2022-12-24T12:31:10.619743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T12:31:13.440324Z","iopub.execute_input":"2022-12-24T12:31:13.440848Z","iopub.status.idle":"2022-12-24T12:31:13.460429Z","shell.execute_reply.started":"2022-12-24T12:31:13.440803Z","shell.execute_reply":"2022-12-24T12:31:13.459124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T12:31:16.451516Z","iopub.execute_input":"2022-12-24T12:31:16.452004Z","iopub.status.idle":"2022-12-24T12:31:16.462075Z","shell.execute_reply.started":"2022-12-24T12:31:16.451966Z","shell.execute_reply":"2022-12-24T12:31:16.46081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission.tail()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T12:31:16.744889Z","iopub.execute_input":"2022-12-24T12:31:16.746235Z","iopub.status.idle":"2022-12-24T12:31:16.755302Z","shell.execute_reply.started":"2022-12-24T12:31:16.746183Z","shell.execute_reply":"2022-12-24T12:31:16.754035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split_dataframes[4].tail()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T12:31:25.306135Z","iopub.execute_input":"2022-12-24T12:31:25.306705Z","iopub.status.idle":"2022-12-24T12:31:25.323338Z","shell.execute_reply.started":"2022-12-24T12:31:25.306658Z","shell.execute_reply":"2022-12-24T12:31:25.321367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split_dataframes[0].head()","metadata":{"execution":{"iopub.status.busy":"2022-12-24T12:31:34.26777Z","iopub.execute_input":"2022-12-24T12:31:34.268319Z","iopub.status.idle":"2022-12-24T12:31:34.286279Z","shell.execute_reply.started":"2022-12-24T12:31:34.268272Z","shell.execute_reply":"2022-12-24T12:31:34.284726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}