{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"## This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n# #     for filename in filenames:\n# #         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-11-28T13:19:06.754257Z","iopub.execute_input":"2022-11-28T13:19:06.754726Z","iopub.status.idle":"2022-11-28T13:19:06.761359Z","shell.execute_reply.started":"2022-11-28T13:19:06.754693Z","shell.execute_reply":"2022-11-28T13:19:06.759792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom glob import glob\nfrom pathlib import Path\nimport sys\n# Local module to simplify plotting\nfrom typing import TYPE_CHECKING, Iterable, Optional\nimport logging\nfrom tqdm import tqdm\nimport gc\nimport h5py","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-11-28T13:19:06.763859Z","iopub.execute_input":"2022-11-28T13:19:06.764256Z","iopub.status.idle":"2022-11-28T13:19:06.983281Z","shell.execute_reply.started":"2022-11-28T13:19:06.764222Z","shell.execute_reply":"2022-11-28T13:19:06.981926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rng = np.random.default_rng()\nimport random","metadata":{"execution":{"iopub.status.busy":"2022-11-28T13:19:06.98765Z","iopub.execute_input":"2022-11-28T13:19:06.988043Z","iopub.status.idle":"2022-11-28T13:19:06.995035Z","shell.execute_reply.started":"2022-11-28T13:19:06.988008Z","shell.execute_reply":"2022-11-28T13:19:06.993111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-11-28T13:19:06.997934Z","iopub.execute_input":"2022-11-28T13:19:06.998358Z","iopub.status.idle":"2022-11-28T13:19:07.006599Z","shell.execute_reply.started":"2022-11-28T13:19:06.998323Z","shell.execute_reply":"2022-11-28T13:19:07.005627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom matplotlib import colors\nfrom scipy import stats","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-11-28T13:19:07.007788Z","iopub.execute_input":"2022-11-28T13:19:07.008217Z","iopub.status.idle":"2022-11-28T13:19:07.387514Z","shell.execute_reply.started":"2022-11-28T13:19:07.008191Z","shell.execute_reply":"2022-11-28T13:19:07.385914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install git+https://github.com/PyFstat/PyFstat@python37","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-11-28T13:19:07.388509Z","iopub.execute_input":"2022-11-28T13:19:07.388818Z","iopub.status.idle":"2022-11-28T13:19:50.283143Z","shell.execute_reply.started":"2022-11-28T13:19:07.388792Z","shell.execute_reply":"2022-11-28T13:19:50.282364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from typing import TYPE_CHECKING, Iterable, Optional\nimport logging","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-11-28T13:19:50.284335Z","iopub.execute_input":"2022-11-28T13:19:50.284733Z","iopub.status.idle":"2022-11-28T13:19:50.290133Z","shell.execute_reply.started":"2022-11-28T13:19:50.284697Z","shell.execute_reply":"2022-11-28T13:19:50.289248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pyfstat\nfrom pyfstat.utils import get_sft_as_arrays","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-11-28T13:19:50.291303Z","iopub.execute_input":"2022-11-28T13:19:50.291646Z","iopub.status.idle":"2022-11-28T13:19:53.047988Z","shell.execute_reply.started":"2022-11-28T13:19:50.291594Z","shell.execute_reply":"2022-11-28T13:19:53.046906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## This is a dataset test for G2NET competition. Not the best one, I hope not the worst one. Judge yourself by power spectrograms. The intent is not to generate a simultated gravitational wave but to draw a set of lines hidden in noise. \n## because the writer is smart and keep in the memory the injected signal there is nothing tidy in the code\n","metadata":{}},{"cell_type":"code","source":"\"\"\"\nUtils\n=====\nUtility functions to simplify tutorials.\n\"\"\"\n\n\n#plt.rcParams[\"font.family\"] = \"Arial\"\n#plt.rcParams[\"font.size\"] = 20\n\n\ndef plot_real_imag_spectrograms(timestamps, frequency, fourier_data):\n    fig, axs = plt.subplots(1, 2, figsize=(16, 10))\n\n    for ax in axs:\n        ax.set(xlabel=\"SFT index\", ylabel=\"Frequency [Hz]\")\n\n    time_in_days = (timestamps - timestamps[0]) / 1800\n\n    axs[0].set_title(\"SFT Real part\")\n    c = axs[0].pcolormesh(\n        time_in_days,\n        frequency,\n        fourier_data.real,\n        norm=colors.CenteredNorm(),\n    )\n    fig.colorbar(c, ax=axs[0], orientation=\"horizontal\", label=\"Fourier Amplitude\")\n\n    axs[1].set_title(\"SFT Imaginary part\")\n    c = axs[1].pcolormesh(\n        time_in_days,\n        frequency,\n        fourier_data.imag,\n        norm=colors.CenteredNorm(),\n    )\n\n    fig.colorbar(c, ax=axs[1], orientation=\"horizontal\", label=\"Fourier Amplitude\")\n\n    return fig, axs\n\n\ndef plot_real_imag_spectrograms_with_gaps(timestamps, frequency, fourier_data, Tsft):\n\n    # Fill up gaps with Nans\n    gap_length = timestamps[1:] - (timestamps[:-1] + Tsft)\n\n    gap_data = [fourier_data[:, 0]]\n    gap_timestamps = [timestamps[0]]\n\n    for ind, gap in enumerate(gap_length):\n        if gap > 0:\n            gap_data.append(np.full_like(fourier_data[:, ind], np.nan + 1j * np.nan))\n            gap_timestamps.append(timestamps[ind] + Tsft)\n\n        gap_data.append(fourier_data[:, ind + 1])\n        gap_timestamps.append(timestamps[ind + 1])\n\n    return plot_real_imag_spectrograms(\n        np.hstack(gap_timestamps), frequency, np.vstack(gap_data).T\n    )\n\n\ndef plot_real_imag_histogram(fourier_data, theoretical_stdev=None):\n\n    fig, ax = plt.subplots(figsize=(16, 10))\n    ax.set(xlabel=\"SFT value\", ylabel=\"PDF\", yscale=\"log\")\n\n    ax.hist(\n        fourier_data.real.ravel(),\n        density=True,\n        bins=\"auto\",\n        histtype=\"step\",\n        lw=2,\n        label=\"Real part\",\n    )\n    ax.hist(\n        fourier_data.imag.ravel(),\n        density=True,\n        bins=\"auto\",\n        histtype=\"step\",\n        lw=2,\n        label=\"Imaginary part\",\n    )\n\n    if theoretical_stdev is not None:\n        x = np.linspace(-4 * theoretical_stdev, 4 * theoretical_stdev, 1000)\n        y = stats.norm(scale=theoretical_stdev).pdf(x)\n        ax.plot(x, y, color=\"black\", ls=\"--\", label=\"Gaussian distribution\")\n\n    ax.legend()\n\n    return fig, ax\n\n\ndef plot_amplitude_phase_spectrograms(timestamps, frequency, fourier_data):\n    fig, axs = plt.subplots(1, 2, figsize=(16, 10))\n\n    for ax in axs:\n        ax.set(xlabel=\"SFT index\", ylabel=\"Frequency [Hz]\")\n\n    time_in_days = (timestamps - timestamps[0]) / 1800\n\n    axs[0].set_title(\"SFT absolute value\")\n    c = axs[0].pcolorfast(\n        time_in_days, frequency, np.absolute(fourier_data), norm=colors.Normalize()\n    )\n    fig.colorbar(c, ax=axs[0], orientation=\"horizontal\", label=\"Value\")\n\n    axs[1].set_title(\"SFT phase\")\n    c = axs[1].pcolorfast(\n        time_in_days, frequency, np.angle(fourier_data), norm=colors.CenteredNorm()\n    )\n\n    fig.colorbar(c, ax=axs[1], orientation=\"horizontal\", label=\"Value\")\n\n    return fig, axs","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-11-28T13:19:53.049306Z","iopub.execute_input":"2022-11-28T13:19:53.050152Z","iopub.status.idle":"2022-11-28T13:19:53.070668Z","shell.execute_reply.started":"2022-11-28T13:19:53.050063Z","shell.execute_reply":"2022-11-28T13:19:53.068348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Utility to read hdf5 file\ndef read_data(file: Path):\n    with h5py.File(file, \"r\") as f:\n        filename = file.stem\n        f = f[filename]\n        h1 = f[\"H1\"]\n        l1 = f[\"L1\"]\n        #freq_hz = list(f[\"frequency_Hz\"])\n        ###\n        h1_stft = h1[\"SFTs\"][()]\n        ###\n        #h1_timestamp = h1[\"timestamps_GPS\"][()]\n        h1_timestamp =1\n        # H2 data\n        l1_stft = l1[\"SFTs\"][()]\n        #l1_timestamp = l1[\"timestamps_GPS\"][()]\n        l1_timestamp =2\n        return {\n            \"H1\": [h1_stft, h1_timestamp],\n            \"L1\": [l1_stft, l1_timestamp],\n            #\"freq_hz\": freq_hz\n        }","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-11-28T13:19:53.075411Z","iopub.execute_input":"2022-11-28T13:19:53.075776Z","iopub.status.idle":"2022-11-28T13:19:53.091561Z","shell.execute_reply.started":"2022-11-28T13:19:53.075743Z","shell.execute_reply":"2022-11-28T13:19:53.089522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_up_logger(\n    outdir: Optional[str] = None,\n    label: Optional[str] = \"pyfstat\",\n    log_level: str = \"INFO\",  # FIXME: Requires Python 3.8 Literal[\"CRITICAL\", \"ERROR\", \"WARNING\", \"INFO\", \"DEBUG\"] = \"INFO\",\n    streams: Optional[Iterable[\"io.TextIOWrapper\"]] = (sys.stdout,),\n    append: bool = True,\n) -> logging.Logger:\n#     \"\"\"Add file and stream handlers to the `pyfstat` logger.\n#     Handler names generated from ``streams`` and ``outdir, label``\n#     must be unique and no duplicated handler will be attached by\n#     this function.\n#     Parameters\n#     ----------\n#     outdir:\n#         Path to outdir directory. If ``None``, no file handler will be added.\n#     label:\n#         Label for the file output handler, i.e.\n#         the log file will be called `label.log`.\n#         Required, in conjunction with ``outdir``, to add a file handler.\n#         Ignored otherwise.\n#     log_level:\n#         Level of logging. This level is imposed on the logger itself and\n#         *every single handler* attached to it.\n#     streams:\n#         Stream to which logging messages will be passed using a\n#         StreamHandler object. By default, log to ``sys.stdout``.\n#         Other common streams include e.g. ``sys.stderr``.\n#     append:\n#         If ``False``, removes all handlers from the `pyfstat` logger\n#         before adding new ones. This removal is not propagated to\n#         handlers on the `root` logger.\n#     Returns\n#     -------\n#     obj:\n#         Configured instance of the ``logging.Logger`` class.\n#     \"\"\"\n    logger = logging.getLogger(\"pyfstat\")\n    logger.setLevel(log_level)\n\n    if not append:\n        for handler in logger.handlers:\n            logger.removeHandler(handler)\n    else:\n        for handler in logger.handlers:\n            handler.setLevel(log_level)\n\n    stream_names = [\n        handler.stream.name\n        for handler in logger.handlers\n        if type(handler) == logging.StreamHandler\n    ]\n    file_names = [\n        handler.baseFilename\n        for handler in logger.handlers\n        if type(handler) == logging.FileHandler\n    ]\n\n    common_formatter = logging.Formatter(\n        \"%(asctime)s.%(msecs)03d %(name)s %(levelname)-8s: %(message)s\",\n        datefmt=\"%y-%m-%d %H:%M:%S\",  # intended to match LALSuite's format\n    )\n\n    for stream in streams or []:\n        if stream.name in stream_names:\n            continue\n        stream_handler = logging.StreamHandler(stream)\n        stream_handler.setFormatter(common_formatter)\n        stream_handler.setLevel(log_level)\n        logger.addHandler(stream_handler)\n\n    if label and outdir:\n        os.makedirs(outdir, exist_ok=True)\n        log_file = os.path.join(outdir, f\"{label}.log\")\n\n        if log_file not in file_names:\n\n            file_handler = logging.FileHandler(log_file)\n            file_handler.setFormatter(common_formatter)\n            file_handler.setLevel(log_level)\n            logger.addHandler(file_handler)\n\n    return logger","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-11-28T13:19:53.093168Z","iopub.execute_input":"2022-11-28T13:19:53.094175Z","iopub.status.idle":"2022-11-28T13:19:53.109491Z","shell.execute_reply.started":"2022-11-28T13:19:53.094123Z","shell.execute_reply":"2022-11-28T13:19:53.107809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logger = set_up_logger(label=\"0_generating_noise\", log_level=\"INFO\")\n\n%matplotlib inline","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-11-28T13:19:53.111355Z","iopub.execute_input":"2022-11-28T13:19:53.111786Z","iopub.status.idle":"2022-11-28T13:19:53.127266Z","shell.execute_reply.started":"2022-11-28T13:19:53.111749Z","shell.execute_reply":"2022-11-28T13:19:53.126239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Prepare Output Directory","metadata":{}},{"cell_type":"code","source":"noise='/kaggle/working/noise/'\nif not os.path.exists(noise):\n    os.mkdir(noise)\nsignal='/kaggle/working/signal/'\nif not os.path.exists(signal):\n    os.mkdir(signal)","metadata":{"execution":{"iopub.status.busy":"2022-11-28T13:19:53.128259Z","iopub.execute_input":"2022-11-28T13:19:53.128567Z","iopub.status.idle":"2022-11-28T13:19:53.142731Z","shell.execute_reply.started":"2022-11-28T13:19:53.128539Z","shell.execute_reply":"2022-11-28T13:19:53.141426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Save as h5py file function","metadata":{}},{"cell_type":"code","source":"def power_spectrogram2(h1_sft,l1_sft):\n    i=0\n    img = np.empty((3,360,360), dtype=np.float32)\n    for elem in [h1_sft,l1_sft]:\n    #matplotlib.rcParams['image.cmap'] = 'viridis'\n       \n        a=elem[0:360, :4320] * 1e22\n\n        p = 2*(a.real**2 + a.imag**2)/(1800 * 1e24** 2) \n        p = a.real**2 + a.imag**2  # power\n        p /= np.mean(p)  # normalize\n        p = np.mean(p.reshape(360, 360,12), axis=2)\n        img[i] = p\n        img[i]=((img[i])/img[i].mean()).astype(np.float32)\n        \n        plt.figure(figsize=(5,5))\n        plt.imshow(img[i])\n    \n        i=i+1\n    img[2] = (img[0]+img[1])\n    xxx=img.reshape(360,360,3)\n#     plt.figure(figsize=(3,5))\n#     plt.imshow(xxx)\n    return img","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-11-28T13:19:53.143774Z","iopub.execute_input":"2022-11-28T13:19:53.144127Z","iopub.status.idle":"2022-11-28T13:19:53.154151Z","shell.execute_reply.started":"2022-11-28T13:19:53.144095Z","shell.execute_reply":"2022-11-28T13:19:53.152927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#for Slicing uncomment  #temp=temp[0:360,:]\ndef make_h5py_file(frequency,fourier_data,timestamps,i,path=\"/kaggle/working/noise/\",name=\"Gausian_Noise\"):\n#frequency,fourier_data,timestamps,path=\"/kaggle/working/noise/\",name=\"Gausian_Noise\",i        \n        file_path=path+name\n        f = h5py.File(f'{file_path}{i}.hdf5','w')\n        g0 = f.create_group(f\"{name}{i}\")\n        g1 = g0.create_group(\"H1\")\n        g2 = g0.create_group(\"L1\")\n        dset = g0.create_dataset(\"frequency_Hz\", data=frequency)\n        temp=fourier_data['H1']\n        #temp=temp[0:360,:]\n        g1.create_dataset(\"SFTs\",data=temp)\n        g1.create_dataset(\"timestamps_GPS\",data=timestamps['H1'])\n        temp=fourier_data['L1']\n        #temp=temp[0:360,:]\n        g2.create_dataset(\"SFTs\",data=temp)\n        g2.create_dataset(\"timestamps_GPS\",data=timestamps['L1'])\n        f.close()\ndef make_h5py_file_slice(frequency,fourier_data,timestamps,i,path=\"/kaggle/working/noise/\",name=\"Gausian_Noise\"):\n#frequency,fourier_data,timestamps,path=\"/kaggle/working/noise/\",name=\"Gausian_Noise\",i        \n        file_path=path+name\n        f = h5py.File(f'{file_path}{i}.hdf5','w')\n        g0 = f.create_group(f\"{name}{i}\")\n        g1 = g0.create_group(\"H1\")\n        g2 = g0.create_group(\"L1\")\n        dset = g0.create_dataset(\"frequency_Hz\", data=frequency)\n        temp=fourier_data['H1']\n        temp=temp[0:360,:]\n        g1.create_dataset(\"SFTs\",data=temp)\n        g1.create_dataset(\"timestamps_GPS\",data=timestamps['H1'])\n        temp=fourier_data['L1']\n        temp=temp[0:360,:]\n        g2.create_dataset(\"SFTs\",data=temp)\n        g2.create_dataset(\"timestamps_GPS\",data=timestamps['L1'])\n        f.close()\ndef make_h5py_file_bslice(frequency,fourier_data,timestamps,i,path=\"/kaggle/working/noise/\",name=\"Gausian_Noise\"):\n#frequency,fourier_data,timestamps,path=\"/kaggle/working/noise/\",name=\"Gausian_Noise\",i        \n        file_path=path+name\n        f = h5py.File(f'{file_path}{i}.hdf5','w')\n        g0 = f.create_group(f\"{name}{i}\")\n        g1 = g0.create_group(\"H1\")\n        g2 = g0.create_group(\"L1\")\n        dset = g0.create_dataset(\"frequency_Hz\", data=frequency)\n        temp=fourier_data['H1']\n        temp=temp[-360:,:]\n        g1.create_dataset(\"SFTs\",data=temp)\n        g1.create_dataset(\"timestamps_GPS\",data=timestamps['H1'])\n        temp=fourier_data['L1']\n        temp=temp[-360:,:]\n        g2.create_dataset(\"SFTs\",data=temp)\n        g2.create_dataset(\"timestamps_GPS\",data=timestamps['L1'])\n        f.close()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-11-28T13:19:53.155591Z","iopub.execute_input":"2022-11-28T13:19:53.155911Z","iopub.status.idle":"2022-11-28T13:19:53.16968Z","shell.execute_reply.started":"2022-11-28T13:19:53.155884Z","shell.execute_reply":"2022-11-28T13:19:53.1681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Clear output of cell ","metadata":{}},{"cell_type":"code","source":"from IPython.display import clear_output as clear","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-11-28T13:19:53.171012Z","iopub.execute_input":"2022-11-28T13:19:53.171308Z","iopub.status.idle":"2022-11-28T13:19:53.188669Z","shell.execute_reply.started":"2022-11-28T13:19:53.17128Z","shell.execute_reply":"2022-11-28T13:19:53.187022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Gaussian Noise and Signal within ","metadata":{}},{"cell_type":"code","source":"coef_list=[]\npower_list=[]\nsuma=0\n\nfor j in range(9):\n    elem=rng.integers(low=2, high=9)\n    #print(elem)\n    suma=suma+elem\n    coef_list.append(elem*86400)\n    power=rng.integers(low=1, high=5)*1e-23\n    power_list.append(power)","metadata":{"execution":{"iopub.status.busy":"2022-11-28T13:19:53.190672Z","iopub.execute_input":"2022-11-28T13:19:53.191006Z","iopub.status.idle":"2022-11-28T13:19:53.200344Z","shell.execute_reply.started":"2022-11-28T13:19:53.190982Z","shell.execute_reply":"2022-11-28T13:19:53.198692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"non_stationary=235","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:00:27.944838Z","iopub.execute_input":"2022-11-28T14:00:27.945348Z","iopub.status.idle":"2022-11-28T14:00:27.952503Z","shell.execute_reply.started":"2022-11-28T14:00:27.945306Z","shell.execute_reply":"2022-11-28T14:00:27.950377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(range(non_stationary)):\n    try:\n        print('ok here 1')\n        number_list = [1e-23, 2e-23, 3e-23, 4e-23, 5e-23,6e-23,7e-23, 8e-23,9e-23,1e-23,9e-24,1e-23,5e-23,6e-24,4e-23]\n        unu=rng.integers(low=2, high=9)\n        doi=rng.integers(low=2, high=9)\n        trei=rng.integers(low=2, high=9)\n        patru=rng.integers(low=3, high=9)\n        cinci=rng.integers(low=1, high=5)\n        coef_list=[]\n        power_list=[]\n        suma=0\n\n        for j in range(9):\n            elem=rng.integers(low=2, high=9)\n            #print(elem)\n            suma=suma+elem\n            coef_list.append(elem*86400)\n            power=rng.integers(low=1, high=7)*1e-23\n            power_list.append(power)\n        print('ok here suma:',suma)\n        test1=240-suma\n        print('ok here test1',test1)\n        if test1>0:\n            indexul=rng.integers(low=2, high=len(coef_list))\n            coef_list[indexul]=test1*86400\n        segment_lengths=coef_list\n        segment_lengths\n        random.choice(number_list)\n        #segment_sqrtSX = [rng.integers(low=1, high=5) * 1e-23, rng.integers(low=1, high=5)* 1e-23, rng.integers(low=1, high=5) *1e-23,rng.integers(low=1, high=5) * 1e-23, rng.integers(low=1, high=5)* 1e-23, rng.integers(low=1, high=5) * 1e-23,rng.integers(low=1, high=5)* 1e-23, rng.integers(low=1, high=5)*1e-23,rng.integers(low=1, high=5)* 1e-23]\n        segment_sqrtSX =power_list\n        #segment_sqrtSX = [random.choice(number_list), random.choice(number_list), random.choice(number_list),random.choice(number_list), random.choice(number_list), random.choice(number_list),random.choice(number_list), random.choice(number_list),random.choice(number_list)]\n        print(len(segment_sqrtSX))\n        print(len(segment_lengths))\n        rng = np.random.default_rng()\n        rints = rng.integers(low=0, high=90)\n        lints = rng.integers(low=0, high=190)\n        pints = rng.integers(low=180, high=190)\n        signalpower = rng.integers(low=3, high=7)\n        #F0 = np.random.uniform(51,497)\n        F0 = rng.integers(low=51, high=497)\n        F1=  np.random.uniform(1e-12,1e-8)\n        sqrtSX = np.random.uniform(5e-24,5e-23)\n        h0 = np.random.uniform(1e-23,2e-23)\n        cosine=np.random.uniform(0.2,1)\n        phi = np.random.uniform(0,2)*np.pi\n        psi = np.random.uniform(0,1)*np.pi\n        tstart = 1238170021\n        #phi = np.random.uniform(1,-1)*np.pi\n        sft_path = []\n\n        print('pas3')\n        noise_kwargs = {\n                    \"outdir\": \"../kaggle/working/noise/\",\n                    \"tstart\": tstart , # Starting time of the observation [GPS time]\n                    \"duration\": pints * 86400,  # Duration [seconds]\n                    \"detectors\": \"H1,L1\",  # Detector to simulate, in this case LIGO Hanford\n                    \"F0\": F0+ np.random.uniform(-0.15,0.15),  # Central frequency of the band to be generated [Hz]\n                    \"Band\": 0.5,  # Frequency band-width around F0 [Hz]\n                    \"sqrtSX\": sqrtSX,  # Single-sided Amplitude Spectral Density of the noise\n                    \"Tsft\": 3600,  # Fourier transform time duration\n                    \"SFTWindowType\": \"tukey\",  # Window function to compute short Fourier transforms\n                    \"SFTWindowBeta\": 0.001 } # Parameter associated to the window function}\n                    #\"phi\":phi}\n                   # for that signal to fit (including a few extra frequency bins!).\n\n        for segment in range(len(segment_lengths)):\n            noise_kwargs[\"label\"] = f\"segment_{segment}\"\n            noise_kwargs[\"duration\"] = segment_lengths[segment]\n            noise_kwargs[\"sqrtSX\"] = segment_sqrtSX[segment]\n\n            if segment > 0:\n                noise_kwargs[\"tstart\"] += noise_kwargs[\"Tsft\"] + segment_lengths[segment - 1]\n\n            writer = pyfstat.Writer(**noise_kwargs)\n            writer.make_data()\n            sft_path.append(writer.sftfilepath)\n\n        #         if i % 20 == 0:\n        #             clear()\n\n        sft_path = \";\".join(sft_path)  # Concatenate different files using ;\n        frequency1, timestamps1, fourier_data1 = get_sft_as_arrays(sft_path)\n        first_index1 = np.argmin(np.abs(frequency1 - F0+0.02))\n        last_index1 = np.argmin(np.abs(frequency1 - F0-0.08))\n#         frequency = frequency[first_index:last_index+1]\n        frequency1 = frequency1[first_index1:last_index1]\n        fourier_data1 = {key: val[first_index1:last_index1 + 1, :]\n                for key, val in fourier_data1.items()}\n\n        #plot_real_imag_spectrograms(timestamps['H1'], frequency, fourier_data['H1'])\n        print(fourier_data1['H1'].shape)\n        if (fourier_data1['H1'].shape[1]>4319) &  (fourier_data1['H1'].shape[0]>=360):\n            make_h5py_file_slice(frequency1,fourier_data1,timestamps1,i,path=\"/kaggle/working/noise/\",name=\"NonStationary_Noise\")\n        #######################################################################################\n        \n#         noise_kwargs = {\n#                     \"outdir\": \"../kaggle/working/noise/\",\n#                     \"tstart\": tstart , # Starting time of the observation [GPS time]\n#                     \"duration\": pints * 86400,  # Duration [seconds]\n#                     \"detectors\": \"H1,L1\",  # Detector to simulate, in this case LIGO Hanford\n#                     \"F0\": F0+ np.random.uniform(-0.15,0.15),  # Central frequency of the band to be generated [Hz]\n#                     \"Band\": 0.5,  # Frequency band-width around F0 [Hz]\n#                     \"sqrtSX\": sqrtSX,  # Single-sided Amplitude Spectral Density of the noise\n#                     \"Tsft\": 3600,  # Fourier transform time duration\n#                     \"SFTWindowType\": \"tukey\",  # Window function to compute short Fourier transforms\n#                     \"SFTWindowBeta\": 0.001 } # Parameter associated to the window function}\n#                     #\"phi\":phi}\n#                    # for that signal to fit (including a few extra frequency bins!).\n        signal_kwargs = {\n        #             \"noiseSFTs\": writer.sftfilepath,\n                    #\"F0\": F0,\n                    \"F1\": F1,\n                    \"Alpha\": 0.3,\n                    \"Delta\": 0,\n                    #\"h0\": sqrtSX/signalpower,\n                    \"h0\": h0/signalpower,\n                    \"cosi\": cosine,\n                    \"psi\": psi,\n                    \"phi\": phi\n                    }  \n        sft_path = []\n        sft_path_signal=[]\n        for segment in range(len(segment_lengths)):\n            noise_kwargs[\"label\"] = f\"segment_{segment}\"\n            noise_kwargs[\"duration\"] = segment_lengths[segment]\n            noise_kwargs[\"sqrtSX\"] = segment_sqrtSX[segment]\n\n            if segment > 0:\n                noise_kwargs[\"tstart\"] += noise_kwargs[\"Tsft\"] + segment_lengths[segment - 1]\n                signal_kwargs[\"tref\"]= noise_kwargs[\"tstart\"]\n            writer = pyfstat.Writer(**noise_kwargs)\n            writer.make_data()\n            sft_path.append(writer.sftfilepath)\n            signal = pyfstat.Writer(**noise_kwargs, **signal_kwargs)\n            signal.make_data()\n            sft_path_signal.append(signal.sftfilepath)\n        #         if i % 20 == 0:\n        #             clear()\n        \n\n        #####Check the plots\n        #plot_amplitude_phase_spectrograms(timestamps['H1'][:360,:], frequency, fourier_data['H1'])\n        ###############################################################################################\n        print('UN MESAJ IMPORTANT 4320',fourier_data1['H1'].shape[1])\n        print('UN MESAJ IMPORTANT 360',fourier_data1['H1'].shape[0])\n#         #######################################################################################\n#         #####Check the plots\n#         try:\n#             plot_amplitude_phase_spectrograms(timestamps['H1'], frequency, fourier_data['H1'])\n#         except:\n#             print('wrong shape')\n#         ###############################################################################################\n#         frequency, timestamps, fourier_data = get_sft_as_arrays(sft_path)\n#         if (fourier_data['H1'].shape[1]>4319) &  (fourier_data['H1'].shape[0]>=360):\n#             make_h5py_file_slice(frequency,fourier_data,timestamps,i,path=\"/kaggle/working/noise/\",name=\"NonStationary_Noise\")\n        #frequency, timestamps, fourier_data = get_sft_as_arrays(sft_path)\n#         #######################################################################################\n#         #####Check the plots\n#         try:\n#             sliced=fourier_data['H1']\n#             sliced=slice[:360,:]\n#             plot_amplitude_phase_spectrograms(timestamps['H1'], frequency, sliced)\n#         except:\n#             print('wrong shape')\n#         ###############################################################################################\n#         if (fourier_data['H1'].shape[1]>4319) &  (fourier_data['H1'].shape[0]>=360):\n#             make_h5py_file_slice(frequency,fourier_data,timestamps,i,path=\"/kaggle/working/noise/\",name=\"NonStationary_Noise_\")\n        ####3Save signal\n        sft_path_signal = \";\".join(sft_path_signal)  # Concatenate different files using ;\n        frequency, timestamps, fourier_data = get_sft_as_arrays(sft_path_signal)\n        first_index = np.argmin(np.abs(frequency - F0+0.02))\n        last_index = np.argmin(np.abs(frequency - F0-0.08))\n#         frequency = frequency[first_index:last_index+1]\n        frequency = frequency[first_index:last_index]\n        fourier_data = {key: val[first_index:last_index + 1, :]\n                for key, val in fourier_data.items()}\n\n        #plot_real_imag_spectrograms(timestamps['H1'], frequency, fourier_data['H1'])\n        print(\"shape semnal\",fourier_data['H1'].shape)\n        if (fourier_data['H1'].shape[1]>4319) &  (fourier_data['H1'].shape[0]>=360):\n            make_h5py_file_slice(frequency,fourier_data,timestamps,i,path=\"/kaggle/working/signal/\",name=\"NonStationary_NoiseAndSignal_\")\n    except:\n        print('macar am incercat')","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:00:28.555152Z","iopub.execute_input":"2022-11-28T14:00:28.556483Z","iopub.status.idle":"2022-11-28T14:07:24.719641Z","shell.execute_reply.started":"2022-11-28T14:00:28.556449Z","shell.execute_reply":"2022-11-28T14:07:24.718469Z"},"_kg_hide-output":true,"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check some results","metadata":{}},{"cell_type":"code","source":"try:\n        #os.chdir(\"/kaggle/working/noise\")\n        \n        gausian_noise = \"/kaggle/working/signal/\"\n        gausian_noise_list= os.listdir(gausian_noise)\n        #print(gausian_noise_list)\n        file = Path('/kaggle/working/signal/NonStationary_NoiseAndSignal_6.hdf5')\n    #/kaggle/working/NoiseGauss3.hdf5\n        data = read_data(file)\n        h1_sft, h1_ts = data[\"H1\"]\n        l1_sft, l1_ts = data[\"L1\"]\n        print(\"shape of H1 sft:\",h1_sft.shape)\n        plt.figure(figsize=(30,2))\n        plt.pcolormesh(h1_sft.real)\n        plt.colorbar()\n        print(\"shape of L1 sft:\",l1_sft.shape)\n        plt.figure(figsize=(30,2))\n        plt.pcolormesh(l1_sft.real)\n        plt.colorbar()\n        try:\n            power_spectrogram2(h1_sft,l1_sft)\n        except:\n            print(\"check the shape of SFT:SFTWindowBeta, Duration, Band \")\n        #plt.imshow()\n        #/kaggle/working/NoiseGauss3.hdf5\nexcept:\n        print(\"test failed\")","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:14:39.964719Z","iopub.execute_input":"2022-11-28T14:14:39.965136Z","iopub.status.idle":"2022-11-28T14:14:43.674577Z","shell.execute_reply.started":"2022-11-28T14:14:39.965101Z","shell.execute_reply":"2022-11-28T14:14:43.672858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:07:28.686393Z","iopub.execute_input":"2022-11-28T14:07:28.686865Z","iopub.status.idle":"2022-11-28T14:07:28.6911Z","shell.execute_reply.started":"2022-11-28T14:07:28.686832Z","shell.execute_reply":"2022-11-28T14:07:28.690398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# gausian_noise = \"/kaggle/working/noise/\"\n# gausian_noise_list= os.listdir(gausian_noise)\n# print(gausian_noise_list)","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:07:28.693192Z","iopub.execute_input":"2022-11-28T14:07:28.694017Z","iopub.status.idle":"2022-11-28T14:07:28.70853Z","shell.execute_reply.started":"2022-11-28T14:07:28.693987Z","shell.execute_reply":"2022-11-28T14:07:28.707702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# gausian_noise = \"/kaggle/working/signal/\"\n# gausian_noise_list= os.listdir(gausian_noise)\n# print(gausian_noise_list)","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:07:28.709876Z","iopub.execute_input":"2022-11-28T14:07:28.710327Z","iopub.status.idle":"2022-11-28T14:07:28.719018Z","shell.execute_reply.started":"2022-11-28T14:07:28.710301Z","shell.execute_reply":"2022-11-28T14:07:28.717766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n        #os.chdir(\"/kaggle/working/noise\")\n        gausian_noise = \"/kaggle/working/noise/\"\n        gausian_noise_list= os.listdir(gausian_noise)\n        #print(gausian_noise_list)\n        file = Path('/kaggle/working/noise/NonStationary_Noise6.hdf5')\n    #/kaggle/working/NoiseGauss3.hdf5\n        data = read_data(file)\n        h1_sft, h1_ts = data[\"H1\"]\n        l1_sft, l1_ts = data[\"L1\"]\n        print(\"shape of H1 sft:\",h1_sft.shape)\n        plt.figure(figsize=(30,2))\n        plt.pcolormesh(h1_sft.real)\n        plt.colorbar()\n        print(\"shape of L1 sft:\",l1_sft.shape)\n        plt.figure(figsize=(30,2))\n        plt.pcolormesh(l1_sft.real)\n        plt.colorbar()\n        try:\n            power_spectrogram2(h1_sft,l1_sft)\n        except:\n            print(\"check the shape of SFT:SFTWindowBeta, Duration, Band \")\n        #plt.imshow()\n        #/kaggle/working/NoiseGauss3.hdf5\nexcept:\n        print(\"test failed\")","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:14:53.078809Z","iopub.execute_input":"2022-11-28T14:14:53.08046Z","iopub.status.idle":"2022-11-28T14:14:56.830967Z","shell.execute_reply.started":"2022-11-28T14:14:53.080394Z","shell.execute_reply":"2022-11-28T14:14:56.830201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"G2Net: Winning with External Data - acknowledgement to\nhttps://www.kaggle.com/code/crischir/g2net-winning-strategy-with-external-data/edit","metadata":{}},{"cell_type":"code","source":"from scipy import stats","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:07:32.536652Z","iopub.execute_input":"2022-11-28T14:07:32.537038Z","iopub.status.idle":"2022-11-28T14:07:32.541621Z","shell.execute_reply.started":"2022-11-28T14:07:32.537005Z","shell.execute_reply":"2022-11-28T14:07:32.540665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom pathlib import Path\nimport h5py\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom matplotlib import colors\nfrom scipy import stats\nimport os\n\nplt.rcParams[\"font.family\"] = \"serif\"\nplt.rcParams[\"font.size\"] = 20","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:07:32.542757Z","iopub.execute_input":"2022-11-28T14:07:32.543038Z","iopub.status.idle":"2022-11-28T14:07:32.572272Z","shell.execute_reply.started":"2022-11-28T14:07:32.543013Z","shell.execute_reply":"2022-11-28T14:07:32.570275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_data_c(file):\n    file = Path(file)\n    with h5py.File(file, \"r\") as f:\n        filename = file.stem\n        f = f[filename]\n        h1 = f[\"H1\"]\n        l1 = f[\"L1\"]\n        freq_hz = f[\"frequency_Hz\"]\n        \n        h1_stft = h1[\"SFTs\"][()]\n        h1_timestamp = h1[\"timestamps_GPS\"][()]\n        # H2 data\n        l1_stft = l1[\"SFTs\"][()]\n        l1_timestamp = l1[\"timestamps_GPS\"][()]\n        \n        return [h1_stft, h1_timestamp],            [l1_stft, l1_timestamp], np.array(freq_hz)","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:07:32.5744Z","iopub.execute_input":"2022-11-28T14:07:32.574817Z","iopub.status.idle":"2022-11-28T14:07:32.586781Z","shell.execute_reply.started":"2022-11-28T14:07:32.574782Z","shell.execute_reply":"2022-11-28T14:07:32.585482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list=np.random.choice(glob.glob('/kaggle/input/g2net-detecting-continuous-gravitational-waves/test/*'), size=100)\n(h1_sfts, h1_ts), (l1_sfts, l1_ts), freq = read_data_c(list[1])","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:07:32.590326Z","iopub.execute_input":"2022-11-28T14:07:32.590693Z","iopub.status.idle":"2022-11-28T14:07:33.119223Z","shell.execute_reply.started":"2022-11-28T14:07:32.590663Z","shell.execute_reply":"2022-11-28T14:07:33.117561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_stat_test = []\nnp.random.seed(123)\nfor fname in tqdm(np.random.choice(glob.glob('/kaggle/input/g2net-detecting-continuous-gravitational-waves/test/*'), size=100)):\n    (h1_sfts, h1_ts), (l1_sfts, l1_ts), freq = read_data_c(fname)\n    stds = np.array([np.std(v) for v in np.array_split(np.absolute(h1_sfts).astype(np.float64), 10,  axis=1)])    \n    \n    (h1_sfts, h1_ts), (l1_sfts, l1_ts), freq = read_data_c(fname)\n    df_stat_test.append({\n        'fname': fname,\n        'minstd': min(stds),\n        'maxstd': max(stds),\n        'stddiff': max(stds) - min(stds)\n    })\n    \nprint('Test set amplitude STD stats')\ndf_stat_test = pd.DataFrame(df_stat_test)\ndf_stat_test","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:07:33.121444Z","iopub.execute_input":"2022-11-28T14:07:33.122487Z","iopub.status.idle":"2022-11-28T14:08:17.698164Z","shell.execute_reply.started":"2022-11-28T14:07:33.122447Z","shell.execute_reply":"2022-11-28T14:08:17.696946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 5))\nax = df_stat_test.stddiff.plot.hist(bins=100, title='Test data: distribution of variation of noise across time buckets')\nax.axvline(1e-24)","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:08:17.700603Z","iopub.execute_input":"2022-11-28T14:08:17.701112Z","iopub.status.idle":"2022-11-28T14:08:18.438506Z","shell.execute_reply.started":"2022-11-28T14:08:17.701074Z","shell.execute_reply":"2022-11-28T14:08:18.436477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 5))\nax = df_stat_test.minstd.plot.hist(bins=100, title='Test data: distribution of variation of noise across time buckets')\nax.axvline(1e-24)","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:08:18.440406Z","iopub.execute_input":"2022-11-28T14:08:18.440839Z","iopub.status.idle":"2022-11-28T14:08:18.778339Z","shell.execute_reply.started":"2022-11-28T14:08:18.440804Z","shell.execute_reply":"2022-11-28T14:08:18.776416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In the distribution above there is a visibile distinction between a set of samples with <1e-24 and >1e-24 variation of noise STD.\n\n< 1e-24 corresponds to the generated stationary noise\n> 1e-24 is nonstationary noise","metadata":{}},{"cell_type":"code","source":"df_stat_train = []\nnp.random.seed(123)\nfor fname in tqdm(np.random.choice(glob.glob('/kaggle/working/signal/*'), size=7)):\n    (h1_sfts, h1_ts), (l1_sfts, l1_ts), freq = read_data_c(fname)\n    stds = np.array([np.std(v) for v in np.array_split(np.absolute(h1_sfts).astype(np.float64), 10,  axis=1)])    \n    \n    (h1_sfts, h1_ts), (l1_sfts, l1_ts), freq = read_data_c(fname)\n    df_stat_train.append({\n        'fname': fname,\n        'minstd': min(stds),\n        'maxstd': max(stds),\n        'stddiff': max(stds) - min(stds),\n        'size1': (h1_sfts.shape)[1],\n        'size2': (h1_sfts.shape)[0],\n        'len': len(h1_ts)\n    })\n    \nprint('Test set amplitude STD stats')\ndf_stat_train = pd.DataFrame(df_stat_train)\ndf_stat_train","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:08:18.779745Z","iopub.execute_input":"2022-11-28T14:08:18.780086Z","iopub.status.idle":"2022-11-28T14:08:19.062336Z","shell.execute_reply.started":"2022-11-28T14:08:18.780059Z","shell.execute_reply":"2022-11-28T14:08:19.061204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_stat_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:08:19.063914Z","iopub.execute_input":"2022-11-28T14:08:19.064272Z","iopub.status.idle":"2022-11-28T14:08:19.100845Z","shell.execute_reply.started":"2022-11-28T14:08:19.064187Z","shell.execute_reply":"2022-11-28T14:08:19.098742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 5))\nax = df_stat_train.stddiff.plot.hist(bins=100, title='Test data: distribution of variation of noise across time buckets')\nax.axvline(1e-24)","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:08:19.104957Z","iopub.execute_input":"2022-11-28T14:08:19.105361Z","iopub.status.idle":"2022-11-28T14:08:19.467685Z","shell.execute_reply.started":"2022-11-28T14:08:19.105329Z","shell.execute_reply":"2022-11-28T14:08:19.466393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_amplitude_spectrogram(timestamps, frequency, fourier_data):\n    \n    ax = plt.subplot(1, 2, 1)\n    ax.set(xlabel=\"SFT index\", ylabel=\"Frequency [Hz]\")\n    time_in_days = (timestamps - timestamps[0]) / 3600 / 24\n    ax.set_title(\"SFT amplitude\")\n    c = ax.pcolorfast(\n        time_in_days, frequency, np.absolute(fourier_data)[:-1, :-1], norm=colors.Normalize()\n    )\n    \n    ax = plt.subplot(1, 2, 2)\n    \n    noise_levels = np.std(np.absolute(fourier_data).astype(np.float64), axis=0)\n    \n    ax.plot(noise_levels)\n    ax.set_title('SFT Amplitude STDDEV')\n    ","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:08:19.470226Z","iopub.execute_input":"2022-11-28T14:08:19.470745Z","iopub.status.idle":"2022-11-28T14:08:19.479691Z","shell.execute_reply.started":"2022-11-28T14:08:19.470699Z","shell.execute_reply":"2022-11-28T14:08:19.4776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"font = {'family' : 'normal',\n        'weight' : 'bold',\n        'size'   : 12}\n\nplt.rc('font', **font)","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:08:19.481689Z","iopub.execute_input":"2022-11-28T14:08:19.4825Z","iopub.status.idle":"2022-11-28T14:08:19.497505Z","shell.execute_reply.started":"2022-11-28T14:08:19.48246Z","shell.execute_reply":"2022-11-28T14:08:19.496244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, (_, r) in enumerate(df_stat_test.query('stddiff < 1e-24').head(4).iterrows()):\n    (h1_sfts, h1_ts), (l1_sfts, l1_ts), freq = read_data_c(r.fname)\n    plt.figure(figsize=(15, 7))\n    plt.suptitle(f'Competition test sample: {os.path.basename(r.fname)}')\n    plot_amplitude_spectrogram(\n        h1_ts, np.array(freq), h1_sfts\n    )","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:08:19.499487Z","iopub.execute_input":"2022-11-28T14:08:19.49998Z","iopub.status.idle":"2022-11-28T14:08:22.426733Z","shell.execute_reply.started":"2022-11-28T14:08:19.499938Z","shell.execute_reply":"2022-11-28T14:08:22.424695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, (_, r) in enumerate(df_stat_train.query('stddiff < 1e-24').head(3).iterrows()):\n    print(fname)\n    (h1_sfts, h1_ts), (l1_sfts, l1_ts), freq = read_data_c(r.fname)\n    plt.figure(figsize=(16, 7))\n    plt.suptitle(f'Generated test sample: {os.path.basename(r.fname)}')\n    plot_amplitude_spectrogram(\n        h1_ts, np.array(freq), h1_sfts\n    )","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:19:32.304801Z","iopub.execute_input":"2022-11-28T14:19:32.305211Z","iopub.status.idle":"2022-11-28T14:19:32.315153Z","shell.execute_reply.started":"2022-11-28T14:19:32.305177Z","shell.execute_reply":"2022-11-28T14:19:32.314135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, (_, r) in enumerate(df_stat_train.query('stddiff > 1e-24').head(3).iterrows()):\n    (h1_sfts, h1_ts), (l1_sfts, l1_ts), freq = read_data_c(r.fname)\n    print(r.fname)\n    plt.figure(figsize=(15, 7))\n    plt.suptitle(f'Generated test sample: {os.path.basename(r.fname)}')\n    print(h1_ts.shape,freq.shape,h1_sfts.shape)\n    plot_amplitude_spectrogram(\n        h1_ts, np.array(freq), h1_sfts\n    )","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:08:22.445137Z","iopub.execute_input":"2022-11-28T14:08:22.44563Z","iopub.status.idle":"2022-11-28T14:08:23.786528Z","shell.execute_reply.started":"2022-11-28T14:08:22.445521Z","shell.execute_reply":"2022-11-28T14:08:23.784306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_stat_noise = []\nnp.random.seed(123)\nfor fname in tqdm(np.random.choice(glob.glob('/kaggle/working/noise/*'), size=10)):\n    (h1_sfts, h1_ts), (l1_sfts, l1_ts), freq = read_data_c(fname)\n    stds = np.array([np.std(v) for v in np.array_split(np.absolute(h1_sfts).astype(np.float64), 10,  axis=1)])    \n    \n    (h1_sfts, h1_ts), (l1_sfts, l1_ts), freq = read_data_c(fname)\n    df_stat_noise.append({\n        'fname': fname,\n        'minstd': min(stds),\n        'maxstd': max(stds),\n        'stddiff': max(stds) - min(stds),\n        'size1': (h1_sfts.shape)[1],\n        'size2': (h1_sfts.shape)[0],\n        'size3': (h1_ts.shape),\n        'size4': (freq.shape)[0],\n        'len': len(h1_ts)\n    })\n    \nprint('Test set amplitude STD stats')\ndf_stat_noise = pd.DataFrame(df_stat_noise)\ndf_stat_noise","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:08:23.787859Z","iopub.execute_input":"2022-11-28T14:08:23.78832Z","iopub.status.idle":"2022-11-28T14:08:24.516435Z","shell.execute_reply.started":"2022-11-28T14:08:23.78828Z","shell.execute_reply":"2022-11-28T14:08:24.514492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_stat_noise.describe()","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:08:24.51809Z","iopub.execute_input":"2022-11-28T14:08:24.518529Z","iopub.status.idle":"2022-11-28T14:08:24.559191Z","shell.execute_reply.started":"2022-11-28T14:08:24.518488Z","shell.execute_reply":"2022-11-28T14:08:24.557386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(h1_sfts, h1_ts), (l1_sfts, l1_ts), freq = read_data_c('/kaggle/working/signal/NonStationary_NoiseAndSignal_6.hdf5')\nprint(h1_ts.shape,freq.shape,h1_sfts.shape)\nplot_amplitude_spectrogram(h1_ts, np.array(freq), h1_sfts)","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:17:36.707179Z","iopub.execute_input":"2022-11-28T14:17:36.707566Z","iopub.status.idle":"2022-11-28T14:17:37.44928Z","shell.execute_reply.started":"2022-11-28T14:17:36.707533Z","shell.execute_reply":"2022-11-28T14:17:37.447842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(h1_sfts, h1_ts), (l1_sfts, l1_ts), freq = read_data_c('/kaggle/working/noise/NonStationary_Noise6.hdf5')\nprint(h1_ts.shape,freq.shape,h1_sfts.shape)\nplot_amplitude_spectrogram(h1_ts, np.array(freq), h1_sfts)","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:17:45.685253Z","iopub.execute_input":"2022-11-28T14:17:45.685716Z","iopub.status.idle":"2022-11-28T14:17:46.007525Z","shell.execute_reply.started":"2022-11-28T14:17:45.685676Z","shell.execute_reply":"2022-11-28T14:17:46.005558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, (_, r) in enumerate(df_stat_train.head(7).iterrows()):\n    (h1_sfts, h1_ts), (l1_sfts, l1_ts), freq = read_data_c(r.fname)\n    print(r.fname)\n    plt.figure(figsize=(15, 7))\n    plt.suptitle(f'Generated test sample: {os.path.basename(r.fname)}')\n    plot_amplitude_spectrogram(\n        h1_ts, np.array(freq), h1_sfts\n    )","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:18:10.000704Z","iopub.execute_input":"2022-11-28T14:18:10.001142Z","iopub.status.idle":"2022-11-28T14:18:14.61612Z","shell.execute_reply.started":"2022-11-28T14:18:10.001108Z","shell.execute_reply":"2022-11-28T14:18:14.614692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, (_, r) in enumerate(df_stat_noise.head(7).iterrows()):\n    (h1_sfts, h1_ts), (l1_sfts, l1_ts), freq = read_data_c(r.fname)\n    print(r.fname)\n    plt.figure(figsize=(15, 7))\n    plt.suptitle(f'Generated test sample: {os.path.basename(r.fname)}')\n    plot_amplitude_spectrogram(\n        h1_ts, np.array(freq), h1_sfts\n    )","metadata":{"execution":{"iopub.status.busy":"2022-11-28T14:18:36.343341Z","iopub.execute_input":"2022-11-28T14:18:36.343824Z","iopub.status.idle":"2022-11-28T14:18:39.816548Z","shell.execute_reply.started":"2022-11-28T14:18:36.343785Z","shell.execute_reply":"2022-11-28T14:18:39.81451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}