{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport gc\n\n!pip install h5py\nimport h5py","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/g2net-detecting-continuous-gravitational-waves/train_labels.csv')\ntrain_dir = '../input/g2net-detecting-continuous-gravitational-waves/train/'\ntest = pd.read_csv('../input/g2net-detecting-continuous-gravitational-waves/sample_submission.csv')\ntest_dir = '../input/g2net-detecting-continuous-gravitational-waves/test/'\nrows = []\niiii = 0\nfor id_ in train['id'].values:\n    f = h5py.File(train_dir + id_+'.hdf5', 'r')\n    number_timeframes = f[id_][\"H1\"]['timestamps_GPS'].shape[0]\n    H1_real = np.real(f[id_][\"H1\"]['SFTs'][:])\n    H1_imag = np.imag(f[id_][\"H1\"]['SFTs'][:])\n    H1_abs = np.abs(f[id_][\"H1\"]['SFTs'][:])\n    L1_real = np.real(f[id_][\"L1\"]['SFTs'][:])\n    L1_imag = np.imag(f[id_][\"L1\"]['SFTs'][:])\n    L1_abs = np.abs(f[id_][\"L1\"]['SFTs'][:])\n    frequency = f[id_][\"frequency_Hz\"][:]\n    for i,f in enumerate(frequency):\n        row = [id_, f]\n        for thersh in [2, 2.2, 2.4, 2.6, 2.8, 3, 3.2, 3.4, 3.6, 3.8, 4, 4.2, 4.4, 4.6, 4.8, 5]:\n            row.append(4000*np.sum(H1_real[i,:]>thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(H1_imag[i,:]>thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(H1_abs[i,:]>thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(L1_real[i,:]>thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(L1_imag[i,:]>thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(L1_abs[i,:]>thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(H1_real[i,:]<-thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(H1_imag[i,:]<-thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(L1_real[i,:]<-thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(L1_imag[i,:]<-thersh*10**(-22))/number_timeframes)\n        row.append(train[train['id']==id_]['target'].values[0])\n        rows.append(row)\n    del f, H1_real, H1_imag, H1_abs, L1_real, L1_imag, L1_abs, frequency\n    gc.collect()\n    iiii +=1\n    if(iiii%100 == 0):\n        pd.DataFrame(rows,columns = ['id','freq']+[f'ftr_{i}' for i in range(160)]+['target']).to_csv('train'+str(int(iiii/100))+'.csv',index = False)\n        del rows\n        gc.collect()\n        rows = []\n\nif(len(rows)>0):\n    pd.DataFrame(rows,columns = ['id','freq']+[f'ftr_{i}' for i in range(160)]+['target']).to_csv('train'+str(1+int(iiii/100))+'.csv',index = False)\n    del rows\n    gc.collect()\n    rows = []\n\niiii = 0\nfor id_ in test['id'].values:\n    f = h5py.File(test_dir + id_+'.hdf5', 'r')\n    number_timeframes = f[id_][\"H1\"]['timestamps_GPS'].shape[0]\n    H1_real = np.real(f[id_][\"H1\"]['SFTs'][:])\n    H1_imag = np.imag(f[id_][\"H1\"]['SFTs'][:])\n    H1_abs = np.abs(f[id_][\"H1\"]['SFTs'][:])\n    L1_real = np.real(f[id_][\"L1\"]['SFTs'][:])\n    L1_imag = np.imag(f[id_][\"L1\"]['SFTs'][:])\n    L1_abs = np.abs(f[id_][\"L1\"]['SFTs'][:])\n    frequency = f[id_][\"frequency_Hz\"][:]\n    for i,f in enumerate(frequency):\n        row = [id_, f]\n        for thersh in [2, 2.2, 2.4, 2.6, 2.8, 3, 3.2, 3.4, 3.6, 3.8, 4, 4.2, 4.4, 4.6, 4.8, 5]:\n            row.append(4000*np.sum(H1_real[i,:]>thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(H1_imag[i,:]>thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(H1_abs[i,:]>thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(L1_real[i,:]>thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(L1_imag[i,:]>thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(L1_abs[i,:]>thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(H1_real[i,:]<-thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(H1_imag[i,:]<-thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(L1_real[i,:]<-thersh*10**(-22))/number_timeframes)\n            row.append(4000*np.sum(L1_imag[i,:]<-thersh*10**(-22))/number_timeframes)\n        rows.append(row)\n    del f, H1_real, H1_imag, H1_abs, L1_real, L1_imag, L1_abs, frequency\n    gc.collect()\n    iiii +=1\n    if(iiii%100 == 0):\n        pd.DataFrame(rows,columns = ['id','freq']+[f'ftr_{i}' for i in range(160)]).to_csv('test'+str(int(iiii/100))+'.csv',index = False)\n        del rows\n        gc.collect()\n        rows = []\n\nif(len(rows)>0):\n    pd.DataFrame(rows,columns = ['id','freq']+[f'ftr_{i}' for i in range(160)]).to_csv('test'+str(1+int(iiii/100))+'.csv',index = False)","metadata":{"execution":{"iopub.status.busy":"2022-10-10T14:23:30.203513Z","iopub.execute_input":"2022-10-10T14:23:30.204397Z","iopub.status.idle":"2022-10-10T14:34:19.915677Z","shell.execute_reply.started":"2022-10-10T14:23:30.204355Z","shell.execute_reply":"2022-10-10T14:34:19.912709Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]}]}