{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":101849,"databundleVersionId":12846694,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy.ma as ma\n\nimport numba\nimport numpy as np\nimport pandas as pd\nimport itertools\nimport os\nimport glob \nfrom astropy.stats import sigma_clip\n\nfrom tqdm import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-07T14:57:36.666806Z","iopub.execute_input":"2025-07-07T14:57:36.66711Z","iopub.status.idle":"2025-07-07T14:57:36.672588Z","shell.execute_reply.started":"2025-07-07T14:57:36.667081Z","shell.execute_reply":"2025-07-07T14:57:36.671362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path_folder = '/kaggle/input/ariel-data-challenge-2025/' # This path stays the same\npath_out = '/kaggle/working/data_light_raw/' # Updated to save files permanently\noutput_dir = '/kaggle/working/data_light_raw/' # Updated to save files permanently","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if not os.path.exists(path_out):\n    os.makedirs(path_out)\n    print(f\"Directory {path_out} created.\")\nelse:\n    print(f\"Directory {path_out} already exists.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T08:33:43.698139Z","iopub.execute_input":"2025-07-07T08:33:43.698437Z","iopub.status.idle":"2025-07-07T08:33:43.723988Z","shell.execute_reply.started":"2025-07-07T08:33:43.698408Z","shell.execute_reply":"2025-07-07T08:33:43.723094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"chunks_size = 5 #because if we load all the files together it might cause an memory overload\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T08:33:43.72697Z","iopub.execute_input":"2025-07-07T08:33:43.727261Z","iopub.status.idle":"2025-07-07T08:33:43.748991Z","shell.execute_reply.started":"2025-07-07T08:33:43.727238Z","shell.execute_reply":"2025-07-07T08:33:43.747808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def correct_ADC_convert(signal, gain=0.4369, offset=-1000):\n    \"\"\"Restores the full dynamic range of the signal according to the competition rules.\"\"\"\n    # Convert to float for calculations\n    signal = signal.astype(np.float64)\n    \n    # 1. Multiply by gain\n    signal *= gain\n    \n    # 2. Add offset\n    signal += offset\n    \n    return signal","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T08:33:43.750281Z","iopub.execute_input":"2025-07-07T08:33:43.750611Z","iopub.status.idle":"2025-07-07T08:33:43.77301Z","shell.execute_reply.started":"2025-07-07T08:33:43.750562Z","shell.execute_reply":"2025-07-07T08:33:43.771688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef mask_hot_dead(signal, dead, dark):\n    \"\"\"\n    Masks hot and dead pixels in a 4D signal array.\n    \"\"\"\n    # --- THIS IS THE FIX ---\n    # Ensure the input signal is a masked array so it has a .mask attribute\n    signal = ma.asarray(signal)\n\n    # Find hot pixels from the 2D dark frame\n    hot_mask_2d = sigma_clip(dark, sigma=5, maxiters=5).mask\n    \n    # Combine the 2D hot and dead pixel maps\n    combined_mask_2d = dead | hot_mask_2d\n    \n    # Directly update the signal's mask using broadcasting.\n    # np.ma.mask_or safely combines the old and new masks.\n    new_mask = combined_mask_2d[np.newaxis, np.newaxis, :, :]\n    signal.mask = ma.mask_or(signal.mask, new_mask)\n    \n    return signal","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T08:33:43.774054Z","iopub.execute_input":"2025-07-07T08:33:43.774386Z","iopub.status.idle":"2025-07-07T08:33:43.802939Z","shell.execute_reply.started":"2025-07-07T08:33:43.774355Z","shell.execute_reply":"2025-07-07T08:33:43.801516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@numba.njit\ndef apply_linear_corr(linear_corr, clean_signal):\n    \"\"\"\n    Applies a unique polynomial linearity correction to each pixel.\n    This version is optimized for speed using Numba.\n    \"\"\"\n    # Create a copy to avoid changing the original input array\n    corrected_signal = np.copy(clean_signal)\n\n    # Loop through each pixel (Numba makes these loops extremely fast)\n    for x in range(clean_signal.shape[1]):\n        for y in range(clean_signal.shape[2]):\n            # Loop through the time dimension for each pixel\n            for t in range(clean_signal.shape[0]):\n                \n                # This manually evaluates the polynomial for the current pixel.\n                # It's a fast implementation of what np.poly1d does.\n                \n                # Get the coefficients and the value for the current pixel\n                coeffs = linear_corr[:, x, y]\n                val = clean_signal[t, x, y]\n                \n                # Apply the polynomial\n                result = coeffs[0]\n                for i in range(1, len(coeffs)):\n                    result = result * val + coeffs[i]\n                \n                clean_signal[t, x, y] = result\n                \n    return clean_signal","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T08:33:43.804485Z","iopub.execute_input":"2025-07-07T08:33:43.805058Z","iopub.status.idle":"2025-07-07T08:33:43.94286Z","shell.execute_reply.started":"2025-07-07T08:33:43.804908Z","shell.execute_reply":"2025-07-07T08:33:43.941673Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef clean_dark(signal, dead, dark, dt):\n    \"\"\"\n    Subtracts a time-scaled dark frame from a signal using broadcasting.\n    \"\"\"\n    # 1. Mask the dead pixels within the dark frame itself.\n    dark_masked = np.ma.masked_where(dead, dark)\n\n    # 2. Use broadcasting to subtract the scaled dark frame.\n    #    NumPy automatically \"stretches\" the 2D dark_masked and 1D dt arrays\n    #    to match the 3D signal's shape during the calculation.\n    corrected_signal = signal - dark_masked[np.newaxis, :, :] * dt[:, np.newaxis, np.newaxis]\n\n    return signal","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T08:33:43.944095Z","iopub.execute_input":"2025-07-07T08:33:43.944643Z","iopub.status.idle":"2025-07-07T08:33:43.951929Z","shell.execute_reply.started":"2025-07-07T08:33:43.944583Z","shell.execute_reply":"2025-07-07T08:33:43.950395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_cds(signal):\n    cds = signal[:,1::2,:,:] - signal[:,::2,:,:]\n    return cds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T08:33:43.952838Z","iopub.execute_input":"2025-07-07T08:33:43.953139Z","iopub.status.idle":"2025-07-07T08:33:43.980448Z","shell.execute_reply.started":"2025-07-07T08:33:43.953113Z","shell.execute_reply":"2025-07-07T08:33:43.979022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def bin_obs(cds_signal, binning):\n    \"\"\"\n    Optimized time binning using reshape and sum.\n    This version handles cases where the time axis is not perfectly divisible.\n    \"\"\"\n    # Get the original shape\n    num_planets, num_frames, height, width = cds_signal.shape\n    \n    # --- THIS IS THE FIX ---\n    # Calculate the largest length that is perfectly divisible by the binning factor\n    new_length = (num_frames // binning) * binning\n    \n    # Trim the array to that new length, discarding the extra frames at the end\n    trimmed_signal = cds_signal[:, :new_length, :, :]\n\n    # Calculate the new number of frames after binning\n    new_num_frames = trimmed_signal.shape[1] // binning\n    \n    # Reshape the *trimmed* signal, which will now work correctly\n    cds_binned = trimmed_signal.reshape(num_planets, \n                                           new_num_frames, \n                                           binning, \n                                           height, \n                                           width).sum(axis=2)\n    \n    return cds_binned","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T08:33:43.981725Z","iopub.execute_input":"2025-07-07T08:33:43.982451Z","iopub.status.idle":"2025-07-07T08:33:44.005728Z","shell.execute_reply.started":"2025-07-07T08:33:43.982421Z","shell.execute_reply":"2025-07-07T08:33:44.004908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def correct_flat_field(signal, flat, dead):\n    \"\"\"\n    Optimized flat-field correction using broadcasting.\n    \"\"\"\n    # 1. Mask the dead pixels within the flat frame.\n    # The .transpose() lines are removed.\n    flat_masked = ma.masked_where(dead, flat)\n\n    # 2. Use broadcasting to divide the signal by the 2D flat field.\n    # This adds new axes to the flat_masked array to align it with the\n    # 4D signal array shape: (chunk, time, height, width)\n    return signal / flat_masked[np.newaxis, np.newaxis, :, :]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T08:33:44.006657Z","iopub.execute_input":"2025-07-07T08:33:44.006984Z","iopub.status.idle":"2025-07-07T08:33:44.043478Z","shell.execute_reply.started":"2025-07-07T08:33:44.00696Z","shell.execute_reply":"2025-07-07T08:33:44.042469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_index(files,chunks_size ):\n    index = []\n    for file in files:\n        file_name = file.split('/')[-1]\n        if file_name.split('_')[0] == 'AIRS-CH0' and file_name.split('_')[1] == 'signal' and file_name.split('_')[2] == '0.parquet':\n            file_index = os.path.basename(os.path.dirname(file))\n            index.append(int(file_index))\n    index = np.array(index)\n    index = np.sort(index) \n    # credit to DennisSakva\n    index=np.array_split(index, len(index)//chunks_size)\n    \n    return index","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T08:33:44.044226Z","iopub.execute_input":"2025-07-07T08:33:44.04453Z","iopub.status.idle":"2025-07-07T08:33:44.070268Z","shell.execute_reply.started":"2025-07-07T08:33:44.044501Z","shell.execute_reply":"2025-07-07T08:33:44.069177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"files = glob.glob(os.path.join(path_folder, 'train/*/*'))\nindex = get_index(files, chunks_size) \naxis_info = pd.read_parquet(os.path.join(path_folder, 'axis_info.parquet'))\n\nDO_MASK = True\nDO_THE_NL_CORR = False # Linearity correction is slow and complex, skipped for this baseline\nDO_DARK = True\nDO_FLAT = True\nTIME_BINNING = True\n\ncut_inf, cut_sup = 39, 321\nl = cut_sup - cut_inf\n\n# =============================================================================\n# 3. MAIN PROCESSING LOOP\n# =============================================================================\nfor n, index_chunk in enumerate(tqdm(index)):\n    \n    # --- 1. EFFICIENTLY LOAD ALL DATA FOR THE CHUNK ---\n    airs_signals = [pd.read_parquet(os.path.join(path_folder,f'train/{pid}/AIRS-CH0_signal_0.parquet')).values.reshape(-1, 32, 356) for pid in index_chunk]\n    fgs1_signals = [pd.read_parquet(os.path.join(path_folder,f'train/{pid}/FGS1_signal_0.parquet')).values.reshape(-1, 32, 32) for pid in index_chunk]\n    \n    AIRS_CH0_clean = np.stack(airs_signals).astype(np.float64)[:, :, :, cut_inf:cut_sup]\n    FGS1_clean = np.stack(fgs1_signals).astype(np.float64)\n    \n    # --- Load calibration files ONCE per chunk ---\n    first_pid = index_chunk[0]\n    flat_airs = pd.read_parquet(os.path.join(path_folder,f'train/{first_pid}/AIRS-CH0_calibration_0/flat.parquet')).values.astype(np.float64).reshape((32, 356))[:, cut_inf:cut_sup]\n    dark_airs = pd.read_parquet(os.path.join(path_folder,f'train/{first_pid}/AIRS-CH0_calibration_0/dark.parquet')).values.astype(np.float64).reshape((32, 356))[:, cut_inf:cut_sup]\n    \n    # *** THIS IS THE FIX FOR THE TypeError ***\n    dead_airs = pd.read_parquet(os.path.join(path_folder,f'train/{first_pid}/AIRS-CH0_calibration_0/dead.parquet')).values.reshape((32, 356))[:, cut_inf:cut_sup] > 0\n\n    flat_fgs = pd.read_parquet(os.path.join(path_folder,f'train/{first_pid}/FGS1_calibration_0/flat.parquet')).values.astype(np.float64).reshape((32, 32))\n    dark_fgs = pd.read_parquet(os.path.join(path_folder,f'train/{first_pid}/FGS1_calibration_0/dark.parquet')).values.astype(np.float64).reshape((32, 32))\n    \n    # *** THIS IS THE FIX FOR THE TypeError ***\n    dead_fgs = pd.read_parquet(os.path.join(path_folder,f'train/{first_pid}/FGS1_calibration_0/dead.parquet')).values.reshape((32, 32)) > 0\n\n    # --- 2. APPLY VECTORIZED CALIBRATIONS TO THE ENTIRE CHUNK ---\n    \n    AIRS_CH0_clean = correct_ADC_convert(AIRS_CH0_clean)\n    FGS1_clean = correct_ADC_convert(FGS1_clean)\n    \n    dt_airs = axis_info['AIRS-CH0-integration_time'].dropna().values\n    dt_airs[1::2] += 0.1\n    dt_fgs1 = np.ones(FGS1_clean.shape[1]) * 0.1\n    dt_fgs1[1::2] += 0.1\n\n    if DO_MASK:\n        AIRS_CH0_clean = mask_hot_dead(AIRS_CH0_clean, dead_airs, dark_airs)\n        FGS1_clean = mask_hot_dead(FGS1_clean, dead_fgs, dark_fgs)\n    \n    if DO_DARK:\n        # Note: Broadcasting dt requires careful reshaping for the 4D array\n        dt_airs_b = dt_airs.reshape(1, -1, 1, 1)\n        dt_fgs1_b = dt_fgs1.reshape(1, -1, 1, 1)\n        AIRS_CH0_clean = AIRS_CH0_clean - dark_airs[np.newaxis, np.newaxis, :, :] * dt_airs_b\n        FGS1_clean = FGS1_clean - dark_fgs[np.newaxis, np.newaxis, :, :] * dt_fgs1_b\n\n    # --- 3. POST-CALIBRATION PROCESSING ---\n    \n    AIRS_cds = get_cds(AIRS_CH0_clean)\n    FGS1_cds = get_cds(FGS1_clean)\n    del AIRS_CH0_clean, FGS1_clean\n\n    if TIME_BINNING:\n        AIRS_cds_binned = bin_obs(AIRS_cds, binning=30)\n        FGS1_cds_binned = bin_obs(FGS1_cds, binning=30*12)\n    else:\n        AIRS_cds_binned = AIRS_cds\n        FGS1_cds_binned = FGS1_cds\n    del AIRS_cds, FGS1_cds\n\n    if DO_FLAT:\n        AIRS_cds_binned = correct_flat_field(AIRS_cds_binned, flat_airs, dead_airs)\n        FGS1_cds_binned = correct_flat_field(FGS1_cds_binned, flat_fgs, dead_fgs)\n    \n    # --- 4. SAVE THE PROCESSED CHUNK ---\n  # Use the .filled() method to replace masked values with np.nan before saving\n    np.save(os.path.join(path_out, f'AIRS_clean_train_{n}.npy'), AIRS_cds_binned.filled(np.nan))\n    np.save(os.path.join(path_out, f'FGS1_train_{n}.npy'), FGS1_cds_binned.filled(np.nan))\n    del AIRS_cds_binned, FGS1_cds_binned","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-07T08:33:44.073669Z","iopub.execute_input":"2025-07-07T08:33:44.074041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_chunk_number(filepath):\n    \"\"\"Extracts the integer number from a chunked filename.\"\"\"\n    return int(filepath.split('_')[-1].split('.')[0])\ndef create_features(airs_npy_files, fgs1_npy_files, labels_df, chunk_size):\n    \"\"\"\n    This function takes paths to calibrated AIRS and FGS1 .npy files\n    and calculates features, processing in batches.\n    \"\"\"\n    all_feature_chunks = []\n    \n    # A helper function to extract the number from the filename\n    def get_chunk_number(filepath):\n        return int(filepath.split('_')[-1].split('.')[0])\n    \n    # Sort the file lists numerically to ensure order\n    airs_npy_files = sorted(airs_npy_files, key=get_chunk_number)\n    fgs1_npy_files = sorted(fgs1_npy_files, key=get_chunk_number)\n    \n    print(\"Processing calibrated data to create features...\")\n    for i in tqdm(range(len(airs_npy_files))):\n        airs_data_chunk = np.load(airs_npy_files[i])\n        fgs1_data_chunk = np.load(fgs1_npy_files[i])\n\n        # Create Light Curves\n        airs_light_curves = np.nansum(airs_data_chunk, axis=(2, 3))\n        fgs1_light_curves = np.nansum(fgs1_data_chunk, axis=(2, 3))\n\n        # Calculate Dip Depth\n        start_transit, end_transit = 75, 105\n        \n        fgs1_out = np.nanmean(np.concatenate([fgs1_light_curves[:, :start_transit], fgs1_light_curves[:, end_transit:]], axis=1), axis=1)\n        fgs1_in = np.nanmean(fgs1_light_curves[:, start_transit:end_transit], axis=1)\n        fgs1_dip_depth = (fgs1_out - fgs1_in) / fgs1_out\n\n        airs_out = np.nanmean(np.concatenate([airs_light_curves[:, :start_transit], airs_light_curves[:, end_transit:]], axis=1), axis=1)\n        airs_in = np.nanmean(airs_light_curves[:, start_transit:end_transit], axis=1)\n        # Corrected this line to divide by 'out' of transit brightness\n        airs_dip_depth = (airs_out - airs_in) / airs_out\n\n        # Get the correct Planet IDs for this chunk\n        start_index = i * chunk_size\n        end_index = start_index + len(airs_data_chunk)\n        chunk_planet_ids = labels_df.iloc[start_index:end_index]['planet_id']\n\n        # Create a feature DataFrame for this chunk\n        chunk_feature_df = pd.DataFrame({\n            'planet_id': chunk_planet_ids,\n            'fgs1_dip_depth': fgs1_dip_depth,\n            'airs_dip_depth': airs_dip_depth\n        })\n        all_feature_chunks.append(chunk_feature_df)\n            \n    # Concatenate the list of DataFrames into one final DataFrame\n    return pd.concat(all_feature_chunks).reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T07:01:18.018973Z","iopub.execute_input":"2025-07-09T07:01:18.019513Z","iopub.status.idle":"2025-07-09T07:01:18.032957Z","shell.execute_reply.started":"2025-07-09T07:01:18.019482Z","shell.execute_reply":"2025-07-09T07:01:18.032201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# PART 1: SETUP (Assumes all helper functions are defined in the cells above)\n# =============================================================================\nimport numpy as np\nimport pandas as pd\nimport os\nimport glob\nfrom tqdm import tqdm\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error\n\n# --- Paths and Parameters ---\n# This path should point to the output of your FIRST notebook (the calibration one)\ncalibrated_train_path = '/kaggle/working/data_light_raw/' \ncompetition_data_path = '/kaggle/input/ariel-data-challenge-2025/'\nCHUNK_SIZE_TRAIN = 5\nCHUNK_SIZE_TEST = 5 # Using a larger chunk size for the test set is fine\n\n# =============================================================================\n# PART 2: FEATURE ENGINEERING AND MODEL TRAINING\n# =============================================================================\n\n# --- Create Training Features ---\nprint(\"--- Creating Training Features ---\")\n# (This assumes the create_features function is defined above)\ntrain_airs_files = glob.glob(os.path.join(calibrated_train_path, 'AIRS_clean_train_*.npy'))\ntrain_fgs1_files = glob.glob(os.path.join(calibrated_train_path, 'FGS1_train_*.npy'))\ntrain_labels = pd.read_csv(os.path.join(competition_data_path, 'train.csv'))\ntrain_labels = train_labels.sort_values(by='planet_id').reset_index(drop=True)\nfeature_df = create_features(train_airs_files, train_fgs1_files, train_labels, CHUNK_SIZE_TRAIN)\nprint(\"Training features created.\")\n\n# --- Train Models ---\nX = feature_df.drop(columns=['planet_id'])\ny = train_labels.drop(columns=['planet_id'])\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\nmodels = []\nuncertainties = []\n\nprint(\"\\n--- Training Models ---\")\nfor i in tqdm(range(y_train.shape[1]), desc=\"Training Models\"):\n    model = lgb.LGBMRegressor(random_state=42, verbose=-1)\n    model.fit(X_train, y_train.iloc[:, i])\n    val_preds = model.predict(X_val)\n    rmse = np.sqrt(mean_squared_error(y_val.iloc[:, i], val_preds))\n    models.append(model)\n    uncertainties.append(rmse)\nuncertainties = np.array(uncertainties)\nprint(\"Model training complete.\")\n\n# =============================================================================\n# PART 3: PREDICTION PIPELINE (for the hidden test set)\n# =============================================================================\n\ntest_path = os.path.join(competition_data_path, 'test')\nif os.path.exists(test_path):\n    print(\"\\n--- Test Data Found: Starting Full Inference Pipeline ---\")\n    \n    # --- A: CALIBRATE RAW TEST DATA ---\n    print(\"Step A: Calibrating raw test data...\")\n    path_folder = competition_data_path\n    path_out = '/kaggle/working/calibrated_test/' # Temporary folder for calibrated test files\n    if not os.path.exists(path_out): os.makedirs(path_out)\n\n    sample_submission = pd.read_csv(os.path.join(path_folder, 'sample_submission.csv'))\n    test_planet_ids = np.sort(sample_submission['planet_id'].unique())\n    num_chunks = len(test_planet_ids) // CHUNK_SIZE_TEST\n    if num_chunks == 0 and len(test_planet_ids) > 0: num_chunks = 1\n    test_index_chunks = np.array_split(test_planet_ids, num_chunks)\n    \n    axis_info = pd.read_parquet(os.path.join(path_folder, 'axis_info.parquet'))\n    DO_MASK, DO_DARK, DO_FLAT, TIME_BINNING = True, True, True, True\n    cut_inf, cut_sup = 39, 321\n    l = cut_sup - cut_inf\n\n    for n, index_chunk in enumerate(tqdm(test_index_chunks, desc=\"Calibrating Test Chunks\")):\n        airs_signals = [pd.read_parquet(os.path.join(path_folder, f'test/{pid}/AIRS-CH0_signal_0.parquet')).values.reshape(-1, 32, 356) for pid in index_chunk]\n        fgs1_signals = [pd.read_parquet(os.path.join(path_folder, f'test/{pid}/FGS1_signal_0.parquet')).values.reshape(-1, 32, 32) for pid in index_chunk]\n        \n        AIRS_CH0_clean = np.stack(airs_signals).astype(np.float64)[:, :, :, cut_inf:cut_sup]\n        FGS1_clean = np.stack(fgs1_signals).astype(np.float64)\n        \n        first_pid = index_chunk[0]\n        flat_airs = pd.read_parquet(os.path.join(path_folder, f'test/{first_pid}/AIRS-CH0_calibration_0/flat.parquet')).values.astype(np.float64).reshape((32, 356))[:, cut_inf:cut_sup]\n        dark_airs = pd.read_parquet(os.path.join(path_folder, f'test/{first_pid}/AIRS-CH0_calibration_0/dark.parquet')).values.astype(np.float64).reshape((32, 356))[:, cut_inf:cut_sup]\n        dead_airs = pd.read_parquet(os.path.join(path_folder, f'test/{first_pid}/AIRS-CH0_calibration_0/dead.parquet')).values.reshape((32, 356))[:, cut_inf:cut_sup] > 0\n        flat_fgs = pd.read_parquet(os.path.join(path_folder, f'test/{first_pid}/FGS1_calibration_0/flat.parquet')).values.astype(np.float64).reshape((32, 32))\n        dark_fgs = pd.read_parquet(os.path.join(path_folder, f'test/{first_pid}/FGS1_calibration_0/dark.parquet')).values.astype(np.float64).reshape((32, 32))\n        dead_fgs = pd.read_parquet(os.path.join(path_folder, f'test/{first_pid}/FGS1_calibration_0/dead.parquet')).values.reshape((32, 32)) > 0\n        \n        AIRS_CH0_clean = correct_ADC_convert(AIRS_CH0_clean)\n        FGS1_clean = correct_ADC_convert(FGS1_clean)\n        \n        dt_airs = axis_info['AIRS-CH0-integration_time'].dropna().values\n        dt_airs[1::2] += 0.1\n        dt_fgs1 = np.ones(FGS1_clean.shape[1]) * 0.1\n        dt_fgs1[1::2] += 0.1\n\n        if DO_MASK:\n            AIRS_CH0_clean = mask_hot_dead(AIRS_CH0_clean, dead_airs, dark_airs)\n            FGS1_clean = mask_hot_dead(FGS1_clean, dead_fgs, dark_fgs)\n        if DO_DARK:\n            AIRS_CH0_clean = clean_dark(AIRS_CH0_clean, dead_airs, dark_airs, dt_airs)\n            FGS1_clean = clean_dark(FGS1_clean, dead_fgs, dark_fgs, dt_fgs1)\n            \n        AIRS_cds = get_cds(AIRS_CH0_clean)\n        FGS1_cds = get_cds(FGS1_clean)\n        del AIRS_CH0_clean, FGS1_clean\n\n        if TIME_BINNING:\n            AIRS_cds_binned = bin_obs(AIRS_cds, binning=30)\n            FGS1_cds_binned = bin_obs(FGS1_cds, binning=30*12)\n        else:\n            AIRS_cds_binned = AIRS_cds\n            FGS1_cds_binned = FGS1_cds\n        del AIRS_cds, FGS1_cds\n\n        if DO_FLAT:\n            AIRS_cds_binned = correct_flat_field(AIRS_cds_binned, flat_airs, dead_airs)\n            FGS1_cds_binned = correct_flat_field(FGS1_cds_binned, flat_fgs, dead_fgs)\n\n        np.save(os.path.join(path_out, f'AIRS_clean_test_{n}.npy'), AIRS_cds_binned.filled(np.nan))\n        np.save(os.path.join(path_out, f'FGS1_clean_test_{n}.npy'), FGS1_cds_binned.filled(np.nan))\n\n    # --- B: CREATE TEST FEATURES ---\n    print(\"\\nStep B: Creating features for test data...\")\n    test_airs_files = glob.glob(os.path.join(path_out, 'AIRS_clean_test_*.npy'))\n    test_fgs1_files = glob.glob(os.path.join(path_out, 'FGS1_clean_test_*.npy'))\n    test_feature_df = create_features(test_airs_files, test_fgs1_files, sample_submission, CHUNK_SIZE_TEST)\n\n    # --- C: MAKE PREDICTIONS & SUBMIT ---\n    print(\"\\nStep C: Making final predictions...\")\n    X_test = test_feature_df.drop(columns=['planet_id'])\n    test_planet_ids = test_feature_df['planet_id']\n    \n    test_predictions = np.zeros((len(X_test), len(models)))\n    for i, model in enumerate(tqdm(models, desc=\"Predicting\")):\n        test_predictions[:, i] = model.predict(X_test)\n\n    test_uncertainties = np.tile(uncertainties, (len(X_test), 1))\n    \n    pred_df = pd.DataFrame(test_predictions, columns=[f'wl_{i+1}' for i in range(283)])\n    unc_df = pd.DataFrame(test_uncertainties, columns=[f'sigma_{i+1}' for i in range(283)])\n    submission_df = pd.DataFrame({'planet_id': test_planet_ids})\n    submission_df = pd.concat([submission_df, pred_df, unc_df], axis=1)\n    \n    submission_df.to_csv('submission.csv', index=False)\n    print(\"\\nSubmission file created successfully!\")\n    \nelse:\n    print(\"\\nTest data not found. This was likely a training run.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T07:01:34.041384Z","iopub.execute_input":"2025-07-09T07:01:34.042491Z","iopub.status.idle":"2025-07-09T07:03:00.155808Z","shell.execute_reply.started":"2025-07-09T07:01:34.042448Z","shell.execute_reply":"2025-07-09T07:03:00.154699Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.read_csv(\"submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T07:03:36.573749Z","iopub.execute_input":"2025-07-09T07:03:36.574554Z","iopub.status.idle":"2025-07-09T07:03:36.62001Z","shell.execute_reply.started":"2025-07-09T07:03:36.574523Z","shell.execute_reply":"2025-07-09T07:03:36.61926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T07:13:35.296968Z","iopub.execute_input":"2025-07-09T07:13:35.29797Z","iopub.status.idle":"2025-07-09T07:13:35.306091Z","shell.execute_reply.started":"2025-07-09T07:13:35.297935Z","shell.execute_reply":"2025-07-09T07:13:35.305276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}