{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70367,"databundleVersionId":9188054,"sourceType":"competition"}],"dockerImageVersionId":30746,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport itertools\nimport os\nimport glob \nfrom astropy.stats import sigma_clip\nimport gc\nimport sys\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:47:38.775256Z","iopub.execute_input":"2024-08-21T13:47:38.777562Z","iopub.status.idle":"2024-08-21T13:47:39.778697Z","shell.execute_reply.started":"2024-08-21T13:47:38.77748Z","shell.execute_reply":"2024-08-21T13:47:39.777424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_folder = '/kaggle/input/ariel-data-challenge-2024/' # path to the folder containing the data\npath_out = '/kaggle/tmp/data_light_raw/' # path to the folder to store the light data\noutput_dir = '/kaggle/tmp/data_light_raw/' # path for the output directory","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-08-21T13:47:39.785597Z","iopub.execute_input":"2024-08-21T13:47:39.786005Z","iopub.status.idle":"2024-08-21T13:47:39.791514Z","shell.execute_reply.started":"2024-08-21T13:47:39.785964Z","shell.execute_reply":"2024-08-21T13:47:39.790278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not os.path.exists(path_out):\n    os.makedirs(path_out)\n    print(f\"Directory {path_out} created.\")\nelse:\n    print(f\"Directory {path_out} already exists.\")","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:47:39.793174Z","iopub.execute_input":"2024-08-21T13:47:39.793602Z","iopub.status.idle":"2024-08-21T13:47:39.806244Z","shell.execute_reply.started":"2024-08-21T13:47:39.793564Z","shell.execute_reply":"2024-08-21T13:47:39.804963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Preprocess functions\n- following the standard function that was there in resources ","metadata":{}},{"cell_type":"code","source":"def ADC_convert(signal, gain, offset):\n    signal = signal.astype(np.float64)\n    signal /= gain\n    signal += offset\n    return signal\n\n\ndef mask_hot_dead(signal, dead, dark):\n    hot = sigma_clip(\n        dark, sigma=5, maxiters=5\n    ).mask\n    hot = np.tile(hot, (signal.shape[0], 1, 1))\n    dead = np.tile(dead, (signal.shape[0], 1, 1))\n    signal = np.ma.masked_where(dead, signal)\n    signal = np.ma.masked_where(hot, signal)\n    return signal\n\ndef apply_linear_corr(linear_corr,clean_signal):\n    linear_corr = np.flip(linear_corr, axis=0)\n    for x, y in itertools.product(\n                range(clean_signal.shape[1]), range(clean_signal.shape[2])\n            ):\n        poli = np.poly1d(linear_corr[:, x, y])\n        clean_signal[:, x, y] = poli(clean_signal[:, x, y])\n    return clean_signal\n    \n\ndef clean_dark(signal, dead, dark, dt):\n\n    dark = np.ma.masked_where(dead, dark)\n    dark = np.tile(dark, (signal.shape[0], 1, 1))\n\n    signal -= dark* dt[:, np.newaxis, np.newaxis]\n    return signal\n\n\ndef get_cds(signal):\n    cds = signal[:,1::2,:,:] - signal[:,::2,:,:]\n    return cds\n\ndef bin_obs(cds_signal,binning):\n    cds_transposed = cds_signal.transpose(0,1,3,2)\n    cds_binned = np.zeros((cds_transposed.shape[0], cds_transposed.shape[1]//binning, cds_transposed.shape[2], cds_transposed.shape[3]))\n    for i in range(cds_transposed.shape[1]//binning):\n        cds_binned[:,i,:,:] = np.sum(cds_transposed[:,i*binning:(i+1)*binning,:,:], axis=1)\n    return cds_binned\n\ndef correct_flat_field(flat,dead, signal):\n    flat = flat.transpose(1, 0)\n    dead = dead.transpose(1, 0)\n    flat = np.ma.masked_where(dead, flat)\n    flat = np.tile(flat, (signal.shape[0], 1, 1))\n    signal = signal / flat\n    return signal","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:47:39.809765Z","iopub.execute_input":"2024-08-21T13:47:39.81028Z","iopub.status.idle":"2024-08-21T13:47:39.829537Z","shell.execute_reply.started":"2024-08-21T13:47:39.810234Z","shell.execute_reply":"2024-08-21T13:47:39.828281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## we will start by getting the index of the training data:\ndef get_index(files ):\n    index = []\n    for file in files :\n        file_name = file.split('/')[-1]\n        if file_name.split('_')[0] == 'AIRS-CH0' and file_name.split('_')[1] == 'signal.parquet':\n            file_index = os.path.basename(os.path.dirname(file))\n            index.append(int(file_index))\n    index = np.array(index)\n    index = np.sort(index) \n#     index = index.reshape(-1, CHUNKS_SIZE)\n    \n    return index","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:47:39.830849Z","iopub.execute_input":"2024-08-21T13:47:39.83124Z","iopub.status.idle":"2024-08-21T13:47:39.852985Z","shell.execute_reply.started":"2024-08-21T13:47:39.831201Z","shell.execute_reply":"2024-08-21T13:47:39.851553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files = glob.glob(os.path.join(path_folder + 'train/', '*/*'))\n\nindex = get_index(files)  ## 48 is hardcoded here but please feel free to remove it if you want to do it for the entire dataset\nplanet =  index[:150] ## will incrase later \n\ntrain_adc_info = pd.read_csv(os.path.join(path_folder, 'train_adc_info.csv'))\ntrain_adc_info = train_adc_info.set_index('planet_id')\naxis_info = pd.read_parquet(os.path.join(path_folder,'axis_info.parquet'))\nDO_MASK = False\nDO_THE_NL_CORR = False\nDO_DARK = False\nDO_FLAT = False\nTIME_BINNING = False\n\nN =len(planet)\ncut_inf, cut_sup = 39, 321\nl = cut_sup - cut_inf","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:47:39.854413Z","iopub.execute_input":"2024-08-21T13:47:39.854792Z","iopub.status.idle":"2024-08-21T13:47:44.068366Z","shell.execute_reply.started":"2024-08-21T13:47:39.854741Z","shell.execute_reply":"2024-08-21T13:47:44.06723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 20  # Choose an appropriate batch size\nnum_batches = N // BATCH_SIZE + (N % BATCH_SIZE != 0)","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:47:44.069776Z","iopub.execute_input":"2024-08-21T13:47:44.070257Z","iopub.status.idle":"2024-08-21T13:47:44.07602Z","shell.execute_reply.started":"2024-08-21T13:47:44.070205Z","shell.execute_reply":"2024-08-21T13:47:44.074673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_batches","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:47:44.077728Z","iopub.execute_input":"2024-08-21T13:47:44.078112Z","iopub.status.idle":"2024-08-21T13:47:44.093731Z","shell.execute_reply.started":"2024-08-21T13:47:44.078074Z","shell.execute_reply":"2024-08-21T13:47:44.092493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.DataFrame(columns=['feature_variable', 'label', 'planet'])","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:47:44.095526Z","iopub.execute_input":"2024-08-21T13:47:44.095941Z","iopub.status.idle":"2024-08-21T13:47:44.106537Z","shell.execute_reply.started":"2024-08-21T13:47:44.095865Z","shell.execute_reply":"2024-08-21T13:47:44.105132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label = pd.read_csv(path_folder+'train_labels.csv')\nlabel.set_index('planet_id', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:47:44.108194Z","iopub.execute_input":"2024-08-21T13:47:44.10858Z","iopub.status.idle":"2024-08-21T13:47:44.230649Z","shell.execute_reply.started":"2024-08-21T13:47:44.108549Z","shell.execute_reply":"2024-08-21T13:47:44.229289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 20 # oose an appropriate batch size\nnum_batches = N // BATCH_SIZE + (N % BATCH_SIZE != 0)\n\n# Allocate the full AIRS_CH0_clean array\n# AIRS_CH0_clean = np.zeros((N, 11250, 32, 282),  dtype=np.float32)\n\nfor batch in range(num_batches):\n    start_idx = batch * BATCH_SIZE\n    end_idx = min((batch + 1) * BATCH_SIZE, N)\n    print(start_idx, end_idx)\n    \n    # Create a batch for temporary storage\n    AIRS_CH0_clean_batch = np.zeros((end_idx - start_idx, 11250, 32, 282), dtype=np.float32)\n    \n    for i, index_ in tqdm(enumerate(planet[start_idx:end_idx])):\n       \n        df = pd.read_parquet(os.path.join(path_folder, f'train/{index_}/AIRS-CH0_signal.parquet'))\n        signal = df.values.astype(np.float32).reshape((df.shape[0], 32, 356))\n        \n        gain = train_adc_info['AIRS-CH0_adc_gain'].loc[index_]\n        offset = train_adc_info['AIRS-CH0_adc_offset'].loc[index_]\n        signal = ADC_convert(signal, gain, offset)\n        \n        dt_airs = axis_info['AIRS-CH0-integration_time'].dropna().values\n        chopped_signal = signal[:, :, cut_inf:cut_sup]\n        \n        del signal, df\n        gc.collect()\n\n        dark = pd.read_parquet(os.path.join(path_folder, f'train/{index_}/AIRS-CH0_calibration/dark.parquet')).values.astype(np.float32).reshape((32, 356))[:, cut_inf:cut_sup]\n        dead_airs = pd.read_parquet(os.path.join(path_folder, f'train/{index_}/AIRS-CH0_calibration/dead.parquet')).values.astype(np.float32).reshape((32, 356))[:, cut_inf:cut_sup]\n        linear_corr = pd.read_parquet(os.path.join(path_folder, f'train/{index_}/AIRS-CH0_calibration/linear_corr.parquet')).values.astype(np.float32).reshape((6, 32, 356))[:, :, cut_inf:cut_sup]\n\n        if DO_MASK:\n            chopped_signal = mask_hot_dead(chopped_signal, dead_airs, dark)\n        \n        AIRS_CH0_clean_batch[i] = chopped_signal\n\n        if DO_THE_NL_CORR:\n            linear_corr_signal = apply_linear_corr(linear_corr, AIRS_CH0_clean_batch[i - start_idx])\n            AIRS_CH0_clean_batch[i - start_idx] = linear_corr_signal\n        \n        if DO_DARK:\n            cleaned_signal = clean_dark(AIRS_CH0_clean_batch[i - start_idx], dead_airs, dark, dt_airs)\n            AIRS_CH0_clean_batch[i - start_idx] = cleaned_signal\n\n        # Free up memory\n        del chopped_signal, dark, dead_airs, linear_corr\n        gc.collect()\n\n    # Copy the processed batch data back into the full AIRS_CH0_clean array\n#     AIRS_CH0_clean[start_idx:end_idx] = AIRS_CH0_clean_batch\n    for i, p in enumerate(planet[start_idx:end_idx]):\n        mean_signal = AIRS_CH0_clean_batch[i].mean(axis=1)\n        label_ = list(label.loc[p].values[1:])\n        split_arrays = [mean_signal[:, i] for i in range(mean_signal.shape[1])]\n        planet_ = [p] * len(split_arrays)\n        new_data = pd.DataFrame({\n        'feature_variable': split_arrays,\n        'label': label_ ,\n        'planet':planet_}) \n        train_data = pd.concat([train_data, new_data], ignore_index=True)\n        del new_data, planet_\n        gc.collect()\n    # Free up batch memory\n    del AIRS_CH0_clean_batch\n    gc.collect()\n\n# Now AIRS_CH0_clean contains the processed data for all planets","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:47:44.232339Z","iopub.execute_input":"2024-08-21T13:47:44.232714Z","iopub.status.idle":"2024-08-21T13:58:37.291481Z","shell.execute_reply.started":"2024-08-21T13:47:44.232682Z","shell.execute_reply":"2024-08-21T13:58:37.290243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# files = glob.glob(os.path.join(path_folder + 'train/', '*/*'))\n\n# index = get_index(files[:2691])  ## 48 is hardcoded here but please feel free to remove it if you want to do it for the entire dataset\n# planet =  index ## will incrase later \n\n# train_adc_info = pd.read_csv(os.path.join(path_folder, 'train_adc_info.csv'))\n# train_adc_info = train_adc_info.set_index('planet_id')\n# axis_info = pd.read_parquet(os.path.join(path_folder,'axis_info.parquet'))\n# DO_MASK = False\n# DO_THE_NL_CORR = False\n# DO_DARK = False\n# DO_FLAT = False\n# TIME_BINNING = False\n\n# N = index.shape[0]\n# cut_inf, cut_sup = 39, 321\n# l = cut_sup - cut_inf\n# AIRS_CH0_clean = np.zeros((len(planet)-650, 11250, 32, 282))\n# print('done AIRS_CHO_chel')\n\n# # FGS1_clean = np.zeros((673, 135000, 32, 32))\n\n# for i, index_ in tqdm(enumerate(planet)):\n#     df = pd.read_parquet(os.path.join(path_folder,f'train/{index_}/AIRS-CH0_signal.parquet'))\n#     signal = df.values.astype(np.float64).reshape((df.shape[0], 32, 356))\n#     gain = train_adc_info['AIRS-CH0_adc_gain'].loc[index_]\n#     offset = train_adc_info['AIRS-CH0_adc_offset'].loc[index_]\n#     signal = ADC_convert(signal, gain, offset)\n#     dt_airs = axis_info['AIRS-CH0-integration_time'].dropna().values\n#     chopped_signal = signal[:, :, cut_inf:cut_sup]\n#     del signal, df\n# #     flat = pd.read_parquet(os.path.join(path_folder,f'train/{index_}/AIRS-CH0_calibration/flat.parquet')).values.astype(np.float64).reshape((32, 356))[:, cut_inf:cut_sup]\n#     dark = pd.read_parquet(os.path.join(path_folder,f'train/{index_}/AIRS-CH0_calibration/dark.parquet')).values.astype(np.float64).reshape((32, 356))[:, cut_inf:cut_sup]\n#     dead_airs = pd.read_parquet(os.path.join(path_folder,f'train/{index_}/AIRS-CH0_calibration/dead.parquet')).values.astype(np.float64).reshape((32, 356))[:, cut_inf:cut_sup]\n#     linear_corr = pd.read_parquet(os.path.join(path_folder,f'train/{index_}/AIRS-CH0_calibration/linear_corr.parquet')).values.astype(np.float64).reshape((6, 32, 356))[:, :, cut_inf:cut_sup]\n#     if DO_MASK:\n#         chopped_signal = mask_hot_dead(chopped_signal, dead_airs, dark)\n#         AIRS_CH0_clean[i] = chopped_signal\n#     else:\n#         AIRS_CH0_clean[i] = chopped_signal\n\n#     if DO_THE_NL_CORR: \n#         linear_corr_signal = apply_linear_corr(linear_corr,AIRS_CH0_clean[i])\n#         AIRS_CH0_clean[i,:, :, :] = linear_corr_signal\n#     del linear_corr\n#     gc.collect()\n\n#     if DO_DARK: \n#         cleaned_signal = clean_dark(AIRS_CH0_clean[i], dead_airs, dark, dt_airs)\n#         AIRS_CH0_clean[i] = cleaned_signal\n#     else: \n#         pass\n#     chopped_signal = None\n#     dark = None\n#     dead_airs = None\n#     cleaned_signal = None\n#     linear_corr_signal = None\n#     del dark, dead_airs, cleaned_signal\n#     gc.collect()  ","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:58:37.293339Z","iopub.execute_input":"2024-08-21T13:58:37.293728Z","iopub.status.idle":"2024-08-21T13:58:37.302469Z","shell.execute_reply.started":"2024-08-21T13:58:37.293695Z","shell.execute_reply":"2024-08-21T13:58:37.301029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for i, p in enumerate(planet):\n#     mean_signal = AIRS_CH0_clean[i].mean(axis=1)\n#     label_ = list(label.loc[p].values[1:])\n#     split_arrays = [mean_signal[:, i] for i in range(mean_signal.shape[1])]\n#     planet_ = [p] * len(split_arrays)\n#     new_data = pd.DataFrame({\n#     'feature_variable': split_arrays,\n#     'label': label_ ,\n#     'planet':planet_}) \n#     train_data = pd.concat([train_data, new_data], ignore_index=True)\n#     del new_data, planet_\n#     gc.collect()\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:58:37.307855Z","iopub.execute_input":"2024-08-21T13:58:37.308692Z","iopub.status.idle":"2024-08-21T13:58:37.319252Z","shell.execute_reply.started":"2024-08-21T13:58:37.308657Z","shell.execute_reply":"2024-08-21T13:58:37.318181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"per_change_list=[]\nfor i in range(len(train_data)):\n    net_signal = train_data['feature_variable'][i][1::2] - train_data['feature_variable'][i][0::2]\n    cum_signal = net_signal.cumsum()\n    window=500\n    smooth_signal = (cum_signal[window:] - cum_signal[:-window]) / window\n    per_change = -((smooth_signal.min()-smooth_signal.max())/smooth_signal.max())*100\n    per_change_list.append(per_change)\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:58:37.32059Z","iopub.execute_input":"2024-08-21T13:58:37.320939Z","iopub.status.idle":"2024-08-21T13:58:42.855201Z","shell.execute_reply.started":"2024-08-21T13:58:37.3209Z","shell.execute_reply":"2024-08-21T13:58:42.854048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['per_change'] = per_change_list","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:58:42.856656Z","iopub.execute_input":"2024-08-21T13:58:42.857022Z","iopub.status.idle":"2024-08-21T13:58:42.879713Z","shell.execute_reply.started":"2024-08-21T13:58:42.856991Z","shell.execute_reply":"2024-08-21T13:58:42.877998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.to_csv('train_data.csv')","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:58:42.881431Z","iopub.execute_input":"2024-08-21T13:58:42.881806Z","iopub.status.idle":"2024-08-21T13:58:49.827173Z","shell.execute_reply.started":"2024-08-21T13:58:42.881772Z","shell.execute_reply":"2024-08-21T13:58:49.826014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(282,564):\n    net_signal = train_data['feature_variable'][i][1::2] - train_data['feature_variable'][i][0::2]\n    cum_signal = net_signal.cumsum()\n    window=500\n    smooth_signal = (cum_signal[window:] - cum_signal[:-window]) / window\n    if i ==335 or i ==527:\n        fig, ax1 = plt.subplots(figsize=(18, 4))\n\n        # Plot on ax1\n        ax1.plot(smooth_signal, label='raw net signal '+ str(i) )\n        ax1.legend()\n\n        # Display the plot\n        plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:58:49.828752Z","iopub.execute_input":"2024-08-21T13:58:49.829273Z","iopub.status.idle":"2024-08-21T13:58:50.457136Z","shell.execute_reply.started":"2024-08-21T13:58:49.829224Z","shell.execute_reply":"2024-08-21T13:58:50.456133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import Ridge\nfrom sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:58:50.458542Z","iopub.execute_input":"2024-08-21T13:58:50.458978Z","iopub.status.idle":"2024-08-21T13:58:51.566794Z","shell.execute_reply.started":"2024-08-21T13:58:50.458939Z","shell.execute_reply":"2024-08-21T13:58:51.565299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the data into features (X) and target (y)\nX = train_data[['per_change']]\ny = train_data['label']\n\n# Perform train-test split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Initialize and train the Ridge regression model\nridge_model = Ridge(alpha=0)  # alpha is the regularization strength\nridge_model.fit(X_train, y_train)\n\n# Make predictions\ny_pred = ridge_model.predict(X_test)\n\n# Calculate metrics\nmse = mean_squared_error(y_test, y_pred)\nmae = mean_absolute_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\n\n# Print metrics\nprint(\"Mean Squared Error (MSE):\", mse)\nprint(\"Mean Absolute Error (MAE):\", mae)\nprint(\"R^2 Score:\", r2)\n\n# Plot the results\nplt.figure(figsize=(12, 6))\n\n# Plot actual vs predicted values\nplt.subplot(1, 2, 1)\nplt.scatter(y_test, y_pred, alpha=0.5)\nplt.xlabel('Actual Values')\nplt.ylabel('Predicted Values')\nplt.title('Actual vs Predicted Values')\n\n# Plot residuals\nresiduals = y_test - y_pred\nplt.subplot(1, 2, 2)\nplt.scatter(y_pred, residuals, alpha=0.5)\nplt.axhline(y=0, color='r', linestyle='--')\nplt.xlabel('Predicted Values')\nplt.ylabel('Residuals')\nplt.title('Residuals vs Predicted Values')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:58:51.568389Z","iopub.execute_input":"2024-08-21T13:58:51.56876Z","iopub.status.idle":"2024-08-21T13:58:52.551105Z","shell.execute_reply.started":"2024-08-21T13:58:51.568728Z","shell.execute_reply":"2024-08-21T13:58:52.549934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## testing the dataset ","metadata":{}},{"cell_type":"code","source":"test_data = pd.DataFrame(columns=['feature_variable', 'planet'])","metadata":{"execution":{"iopub.status.busy":"2024-08-21T13:58:52.552465Z","iopub.execute_input":"2024-08-21T13:58:52.552794Z","iopub.status.idle":"2024-08-21T13:58:52.559673Z","shell.execute_reply.started":"2024-08-21T13:58:52.552764Z","shell.execute_reply":"2024-08-21T13:58:52.558452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files_test = glob.glob(os.path.join(path_folder + 'test/', '*/*'))\nindex_test = get_index(files_test)\nplanet_test =  index_test\ntest_adc_info = pd.read_csv(path_folder+'test_adc_info.csv')\ntest_adc_info = test_adc_info.set_index('planet_id')\n\nDO_MASK = False\nDO_THE_NL_CORR = False\nDO_DARK = False\nDO_FLAT = False\nTIME_BINNING = False\n\nN = len(planet_test)\ncut_inf, cut_sup = 39, 321\nl = cut_sup - cut_inf","metadata":{"execution":{"iopub.status.busy":"2024-08-21T14:07:28.212993Z","iopub.execute_input":"2024-08-21T14:07:28.213464Z","iopub.status.idle":"2024-08-21T14:07:28.228125Z","shell.execute_reply.started":"2024-08-21T14:07:28.213429Z","shell.execute_reply":"2024-08-21T14:07:28.226728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 10 # oose an appropriate batch size\nnum_batches = N // BATCH_SIZE + (N % BATCH_SIZE != 0)\n\n# Allocate the full AIRS_CH0_clean array\n# AIRS_CH0_clean = np.zeros((N, 11250, 32, 282),  dtype=np.float32)\nfor batch in range(num_batches):\n    start_idx = batch * BATCH_SIZE\n    end_idx = min((batch + 1) * BATCH_SIZE, N)\n    print(start_idx, end_idx)\n    \n    # Create a batch for temporary storage\n    AIRS_CH0_clean_batch = np.zeros((end_idx - start_idx, 11250, 32, 282), dtype=np.float32)\n    \n    for i, index_ in tqdm(enumerate(planet_test[start_idx:end_idx])):\n       \n        df = pd.read_parquet(os.path.join(path_folder, f'test/{index_}/AIRS-CH0_signal.parquet'))\n        signal = df.values.astype(np.float32).reshape((df.shape[0], 32, 356))\n        \n        gain = test_adc_info['AIRS-CH0_adc_gain'].loc[index_]\n        offset = test_adc_info['AIRS-CH0_adc_offset'].loc[index_]\n        signal = ADC_convert(signal, gain, offset)\n        \n        dt_airs = axis_info['AIRS-CH0-integration_time'].dropna().values\n        chopped_signal = signal[:, :, cut_inf:cut_sup]\n        \n        del signal, df\n        gc.collect()\n\n        dark = pd.read_parquet(os.path.join(path_folder, f'test/{index_}/AIRS-CH0_calibration/dark.parquet')).values.astype(np.float32).reshape((32, 356))[:, cut_inf:cut_sup]\n        dead_airs = pd.read_parquet(os.path.join(path_folder, f'test/{index_}/AIRS-CH0_calibration/dead.parquet')).values.astype(np.float32).reshape((32, 356))[:, cut_inf:cut_sup]\n        linear_corr = pd.read_parquet(os.path.join(path_folder, f'test/{index_}/AIRS-CH0_calibration/linear_corr.parquet')).values.astype(np.float32).reshape((6, 32, 356))[:, :, cut_inf:cut_sup]\n\n        if DO_MASK:\n            chopped_signal = mask_hot_dead(chopped_signal, dead_airs, dark)\n        \n        AIRS_CH0_clean_batch[i] = chopped_signal\n\n        if DO_THE_NL_CORR:\n            linear_corr_signal = apply_linear_corr(linear_corr, AIRS_CH0_clean_batch[i - start_idx])\n            AIRS_CH0_clean_batch[i - start_idx] = linear_corr_signal\n        \n        if DO_DARK:\n            cleaned_signal = clean_dark(AIRS_CH0_clean_batch[i - start_idx], dead_airs, dark, dt_airs)\n            AIRS_CH0_clean_batch[i - start_idx] = cleaned_signal\n\n        # Free up memory\n        del chopped_signal, dark, dead_airs, linear_corr\n        gc.collect()\n\n    # Copy the processed batch data back into the full AIRS_CH0_clean array\n#     AIRS_CH0_clean[start_idx:end_idx] = AIRS_CH0_clean_batch\n\n    for i, p in enumerate(planet_test[start_idx:end_idx]):\n        mean_signal = AIRS_CH0_clean_batch[i].mean(axis=1)\n    #     net_signal = mean_signal[1::2] - mean_signal[0::2]\n    #     cum_signal = net_signal.cumsum()\n    #     window=80\n    #     smooth_signal = (cum_signal[window:] - cum_signal[:-window]) / window\n    #     label_ = list(label.loc[p].values[1:])\n        split_arrays = [mean_signal[:, i] for i in range(mean_signal.shape[1])]\n        planet_ = [p] * len(split_arrays)\n        new_data = pd.DataFrame({\n        'feature_variable': split_arrays,\n    #     'label': label_ ,\n        'planet':planet_}) \n        test_data = pd.concat([test_data, new_data], ignore_index=True)\n        del new_data, planet_\n        gc.collect()\n    \n    # Free up batch memory\n    del AIRS_CH0_clean_batch\n    gc.collect()\n\n# Now AIRS_CH0_clean contains the processed data for all planets","metadata":{"execution":{"iopub.status.busy":"2024-08-21T14:07:31.60899Z","iopub.execute_input":"2024-08-21T14:07:31.609463Z","iopub.status.idle":"2024-08-21T14:07:38.505989Z","shell.execute_reply.started":"2024-08-21T14:07:31.609426Z","shell.execute_reply":"2024-08-21T14:07:38.504659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"282*2","metadata":{"execution":{"iopub.status.busy":"2024-08-21T14:07:39.078639Z","iopub.execute_input":"2024-08-21T14:07:39.079662Z","iopub.status.idle":"2024-08-21T14:07:39.087846Z","shell.execute_reply.started":"2024-08-21T14:07:39.079616Z","shell.execute_reply":"2024-08-21T14:07:39.086368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# files_test = glob.glob(os.path.join(path_folder + 'test/', '*/*'))\n# index_test = get_index(files_test)\n# planet_test =  index_test\n# test_adc_info = pd.read_csv(path_folder+'test_adc_info.csv')\n# test_adc_info = test_adc_info.set_index('planet_id')\n# # axis_info = pd.read_parquet(os.path.join(path_folder,'axis_info.parquet'))\n# DO_MASK = False\n# DO_THE_NL_CORR = False\n# DO_DARK = False\n# DO_FLAT = False\n# TIME_BINNING = False\n\n# N = index.shape[0]\n# cut_inf, cut_sup = 39, 321\n# l = cut_sup - cut_inf\n# AIRS_CH0_clean = np.zeros((len(planet_test), 11250, 32, 282))\n# # FGS1_clean = np.zeros((673, 135000, 32, 32))\n\n# for i, index_ in tqdm(enumerate(planet_test)):\n#     df = pd.read_parquet(os.path.join(path_folder,f'test/{index_}/AIRS-CH0_signal.parquet'))\n#     signal = df.values.astype(np.float64).reshape((df.shape[0], 32, 356))\n#     gain = test_adc_info['AIRS-CH0_adc_gain'].loc[index_]\n#     offset = test_adc_info['AIRS-CH0_adc_offset'].loc[index_]\n#     signal = ADC_convert(signal, gain, offset)\n# #     dt_airs = axis_info['AIRS-CH0-integration_time'].dropna().values\n#     chopped_signal = signal[:, :, cut_inf:cut_sup]\n#     del signal, df\n# #     flat = pd.read_parquet(os.path.join(path_folder,f'train/{index_}/AIRS-CH0_calibration/flat.parquet')).values.astype(np.float64).reshape((32, 356))[:, cut_inf:cut_sup]\n#     dark = pd.read_parquet(os.path.join(path_folder,f'test/{index_}/AIRS-CH0_calibration/dark.parquet')).values.astype(np.float64).reshape((32, 356))[:, cut_inf:cut_sup]\n#     dead_airs = pd.read_parquet(os.path.join(path_folder,f'test/{index_}/AIRS-CH0_calibration/dead.parquet')).values.astype(np.float64).reshape((32, 356))[:, cut_inf:cut_sup]\n#     linear_corr = pd.read_parquet(os.path.join(path_folder,f'test/{index_}/AIRS-CH0_calibration/linear_corr.parquet')).values.astype(np.float64).reshape((6, 32, 356))[:, :, cut_inf:cut_sup]\n#     if DO_MASK:\n#         chopped_signal = mask_hot_dead(chopped_signal, dead_airs, dark)\n#         AIRS_CH0_clean[i] = chopped_signal\n#     else:\n#         AIRS_CH0_clean[i] = chopped_signal\n\n#     if DO_THE_NL_CORR: \n#         linear_corr_signal = apply_linear_corr(linear_corr,AIRS_CH0_clean[i])\n#         AIRS_CH0_clean[i,:, :, :] = linear_corr_signal\n#     del linear_corr\n#     gc.collect()\n\n#     if DO_DARK: \n#         cleaned_signal = clean_dark(AIRS_CH0_clean[i], dead_airs, dark, dt_airs)\n#         AIRS_CH0_clean[i] = cleaned_signal\n#     else: \n#         pass\n#     chopped_signal = None\n#     dark = None\n#     dead_airs = None\n#     cleaned_signal = None\n#     linear_corr_signal = None\n#     del dark, dead_airs, cleaned_signal\n#     gc.collect()\n\n    \n    \n    \n    \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2024-08-21T14:07:39.94048Z","iopub.execute_input":"2024-08-21T14:07:39.940942Z","iopub.status.idle":"2024-08-21T14:07:39.949421Z","shell.execute_reply.started":"2024-08-21T14:07:39.940894Z","shell.execute_reply":"2024-08-21T14:07:39.948173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# for i, p in enumerate(planet_test):\n#     mean_signal = AIRS_CH0_clean[i].mean(axis=1)\n# #     net_signal = mean_signal[1::2] - mean_signal[0::2]\n# #     cum_signal = net_signal.cumsum()\n# #     window=80\n# #     smooth_signal = (cum_signal[window:] - cum_signal[:-window]) / window\n# #     label_ = list(label.loc[p].values[1:])\n#     split_arrays = [mean_signal[:, i] for i in range(mean_signal.shape[1])]\n#     planet_ = [p] * len(split_arrays)\n#     new_data = pd.DataFrame({\n#     'feature_variable': split_arrays,\n# #     'label': label_ ,\n#     'planet':planet_}) \n#     test_data = pd.concat([test_data, new_data], ignore_index=True)\n#     del new_data, planet_\n#     gc.collect()\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2024-08-21T14:07:40.463815Z","iopub.execute_input":"2024-08-21T14:07:40.464958Z","iopub.status.idle":"2024-08-21T14:07:40.470624Z","shell.execute_reply.started":"2024-08-21T14:07:40.464906Z","shell.execute_reply":"2024-08-21T14:07:40.469205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_feature_variable = []\nper_change_list=[]\nwindow = 500\nfor i in range(len(test_data)):\n    net_signal = test_data['feature_variable'][i][1::2] - test_data['feature_variable'][i][0::2]\n    cum_signal = net_signal.cumsum()\n    window=500\n    smooth_signal = (cum_signal[window:] - cum_signal[:-window]) / window\n    final_feature_variable.append(smooth_signal)\n    per_change = -((smooth_signal.min()-smooth_signal.max())/smooth_signal.max())*100\n    per_change_list.append(per_change)\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2024-08-21T14:07:40.675142Z","iopub.execute_input":"2024-08-21T14:07:40.67557Z","iopub.status.idle":"2024-08-21T14:07:42.650041Z","shell.execute_reply.started":"2024-08-21T14:07:40.675537Z","shell.execute_reply":"2024-08-21T14:07:42.648821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['smooth_signal'] = final_feature_variable\ntest_data['per_change'] = per_change_list","metadata":{"execution":{"iopub.status.busy":"2024-08-21T14:07:42.652268Z","iopub.execute_input":"2024-08-21T14:07:42.652635Z","iopub.status.idle":"2024-08-21T14:07:42.676556Z","shell.execute_reply.started":"2024-08-21T14:07:42.652603Z","shell.execute_reply":"2024-08-21T14:07:42.675208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax1 = plt.subplots(figsize=(18, 4))\n\n# Plot on ax1\nax1.plot(test_data['smooth_signal'][2], label='raw net signal '+ str(i) )\nax1.legend()\n\n# Display the plot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-08-21T14:07:42.678427Z","iopub.execute_input":"2024-08-21T14:07:42.678787Z","iopub.status.idle":"2024-08-21T14:07:42.992271Z","shell.execute_reply.started":"2024-08-21T14:07:42.678757Z","shell.execute_reply":"2024-08-21T14:07:42.990441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = ridge_model.predict(test_data['per_change'].values.reshape(-1,1))","metadata":{"execution":{"iopub.status.busy":"2024-08-21T14:07:42.994978Z","iopub.execute_input":"2024-08-21T14:07:42.996058Z","iopub.status.idle":"2024-08-21T14:07:43.003304Z","shell.execute_reply.started":"2024-08-21T14:07:42.996017Z","shell.execute_reply":"2024-08-21T14:07:43.001957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = ['planet_id'] + [f'wl_{i+1}' for i in range(283)] + [f'sigma_{i+1}' for i in range(283)]\n\n# Creating an empty DataFrame with the specified columns\nsub = pd.DataFrame(columns=columns)","metadata":{"execution":{"iopub.status.busy":"2024-08-21T14:07:43.004815Z","iopub.execute_input":"2024-08-21T14:07:43.005237Z","iopub.status.idle":"2024-08-21T14:07:43.047933Z","shell.execute_reply.started":"2024-08-21T14:07:43.005205Z","shell.execute_reply":"2024-08-21T14:07:43.046587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(planet_test)):\n#     print(i)\n    first_col_value = planet_test[i]\n    second_col_value = 0.002745\n    remaining_values = [0.000266] * (567 - 1 - 1 - 282)\n    all_values = [first_col_value, second_col_value] + y_pred[i*282:(i+1)*282].tolist() + remaining_values\n    temp = pd.DataFrame([all_values], columns=[f'col_{i+1}' for i in range(567)])\n    temp.columns = ['planet_id'] + [f'wl_{i+1}' for i in range(283)] + [f'sigma_{i+1}' for i in range(283)]\n    sub = pd.concat([sub, temp], ignore_index=True)\n    \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2024-08-21T14:07:43.478782Z","iopub.execute_input":"2024-08-21T14:07:43.479905Z","iopub.status.idle":"2024-08-21T14:07:43.507484Z","shell.execute_reply.started":"2024-08-21T14:07:43.479837Z","shell.execute_reply":"2024-08-21T14:07:43.506165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub = sub.set_index('planet_id')","metadata":{"execution":{"iopub.status.busy":"2024-08-21T14:07:44.934749Z","iopub.execute_input":"2024-08-21T14:07:44.935218Z","iopub.status.idle":"2024-08-21T14:07:44.940364Z","shell.execute_reply.started":"2024-08-21T14:07:44.935183Z","shell.execute_reply":"2024-08-21T14:07:44.939088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-08-21T14:08:44.390256Z","iopub.execute_input":"2024-08-21T14:08:44.390721Z","iopub.status.idle":"2024-08-21T14:08:44.40241Z","shell.execute_reply.started":"2024-08-21T14:08:44.390686Z","shell.execute_reply":"2024-08-21T14:08:44.401062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}