{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":101849,"databundleVersionId":13093295,"sourceType":"competition"},{"sourceId":9629432,"sourceType":"datasetVersion","datasetId":5846888}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# install pqdm for parallel processing\n!pip install --no-index --find-links=/kaggle/input/ariel-2024-pqdm pqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-15T08:57:00.801307Z","iopub.execute_input":"2025-08-15T08:57:00.801579Z","iopub.status.idle":"2025-08-15T08:57:06.15639Z","shell.execute_reply.started":"2025-08-15T08:57:00.801547Z","shell.execute_reply":"2025-08-15T08:57:06.155Z"}},"outputs":[{"name":"stdout","text":"Looking in links: /kaggle/input/ariel-2024-pqdm\nProcessing /kaggle/input/ariel-2024-pqdm/pqdm-0.2.0-py2.py3-none-any.whl\nProcessing /kaggle/input/ariel-2024-pqdm/bounded_pool_executor-0.0.3-py3-none-any.whl (from pqdm)\nRequirement already satisfied: tqdm in /usr/local/lib/python3.11/dist-packages (from pqdm) (4.67.1)\nRequirement already satisfied: typing-extensions in /usr/local/lib/python3.11/dist-packages (from pqdm) (4.14.0)\nInstalling collected packages: bounded-pool-executor, pqdm\nSuccessfully installed bounded-pool-executor-0.0.3 pqdm-0.2.0\n","output_type":"stream"}],"execution_count":1},{"cell_type":"code","source":"# Standard imports\nimport os\nimport itertools\nimport pickle\nfrom tqdm import tqdm\nfrom pqdm.threads import pqdm\nimport numpy as np\nimport pandas as pd\nfrom scipy.optimize import curve_fit\nfrom scipy.signal import savgol_filter\nfrom scipy.optimize import minimize\nfrom astropy.stats import sigma_clip\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-15T08:57:06.159176Z","iopub.execute_input":"2025-08-15T08:57:06.159558Z","iopub.status.idle":"2025-08-15T08:57:08.143262Z","shell.execute_reply.started":"2025-08-15T08:57:06.159516Z","shell.execute_reply":"2025-08-15T08:57:08.142322Z"}},"outputs":[],"execution_count":2},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n\naxis = pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2025/axis_info.parquet\")\n\nair_wl = axis['AIRS-CH0-axis2-um'][:356]\n\ntest_wl = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2025/wavelengths.csv\")\n\n\nsubmission_airs_df = test_wl.loc[0, 'wl_2':]\n\n\n\n# This list will store our results\nmapping_results = []\n\n\nfor channel_name, submission_wavelength in submission_airs_df.items():\n    \n    # For the current submission wavelength, find the index of the closest value\n    # in the full 356-pixel raw wavelength grid.\n    differences = np.abs(air_wl - submission_wavelength)\n    closest_raw_pixel_index = differences.idxmin() # idxmin() gives the index of the minimum value\n\n    # Get the corresponding wavelength value from the raw grid\n    closest_raw_wavelength = air_wl.loc[closest_raw_pixel_index]\n\n    # Store the mapping information\n    mapping_results.append({\n        'Submission_Channel': channel_name,\n        'Submission_Wavelength': submission_wavelength,\n        'Closest_Raw_Pixel_Index': closest_raw_pixel_index,\n        'Closest_Raw_Wavelength': closest_raw_wavelength,\n        'Wavelength_Difference': np.abs(submission_wavelength - closest_raw_wavelength)\n    })\n\nmapping_df = pd.DataFrame(mapping_results)\n\n\nprint(mapping_df.head())\n\nprint(mapping_df.tail())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-15T08:57:08.14425Z","iopub.execute_input":"2025-08-15T08:57:08.144708Z","iopub.status.idle":"2025-08-15T08:57:08.504741Z","shell.execute_reply.started":"2025-08-15T08:57:08.144677Z","shell.execute_reply":"2025-08-15T08:57:08.503714Z"}},"outputs":[{"name":"stdout","text":"  Submission_Channel  Submission_Wavelength  Closest_Raw_Pixel_Index  \\\n0               wl_2               1.951761                      320   \n1               wl_3               1.960612                      319   \n2               wl_4               1.969450                      318   \n3               wl_5               1.978273                      317   \n4               wl_6               1.987083                      316   \n\n   Closest_Raw_Wavelength  Wavelength_Difference  \n0                1.951761           1.332268e-15  \n1                1.960612           1.998401e-15  \n2                1.969450           1.554312e-15  \n3                1.978273           1.776357e-15  \n4                1.987083           1.554312e-15  \n    Submission_Channel  Submission_Wavelength  Closest_Raw_Pixel_Index  \\\n277             wl_279               3.875034                       43   \n278             wl_280               3.880055                       42   \n279             wl_281               3.885063                       41   \n280             wl_282               3.890056                       40   \n281             wl_283               3.895036                       39   \n\n     Closest_Raw_Wavelength  Wavelength_Difference  \n277                3.875034           8.881784e-16  \n278                3.880055           4.440892e-16  \n279                3.885063           4.440892e-16  \n280                3.890056           4.440892e-16  \n281                3.895036           0.000000e+00  \n","output_type":"stream"}],"execution_count":3},{"cell_type":"code","source":"class Config:\n    DATA_PATH = '/kaggle/input/ariel-data-challenge-2025'\n    DATASET = 'train'\n    PREPROCESSED_PATH = '/kaggle/input/adc2025s-preprocessed-signals/preprocessed_signal.csv'\n\n    SCALE = 0.93960\n    SIGMA = 0.0009\n\n    CUT_INF = 39\n    CUT_SUP = 321\n\n    SENSOR_CONFIG = {\n        'AIRS-CH0': {\n            'raw_shape': [11250, 32, 356],\n            'calibrated_shape': [1, 32, CUT_SUP - CUT_INF],\n            'linear_corr_shape': (6, 32, 356),\n            'dt_pattern': (0.1, 4.5),\n            'binning': 30\n        },\n        'FGS1': {\n            'raw_shape': [135000, 32, 32],\n            'calibrated_shape': [1, 32, 32],\n            'linear_corr_shape': (6, 32, 32),\n            'dt_pattern': (0.1, 0.1),\n            'binning': 30 * 12\n        }\n    }\n\n    MODEL_PHASE_DETECTION_SLICE = slice(30, 140)\n    MODEL_OPTIMIZATION_DELTA = 7\n    MODEL_POLYNOMIAL_DEGREE = 3\n\n    N_JOBS = 4\n\n    # autoencoder / nmf\n    AE_ENCODING_DIM = 4\n    NMF_COMPONENTS = 5\n\n    @classmethod\n    def get_planet_ids(cls):\n        df = pd.read_csv(f'{cls.DATA_PATH}/{cls.DATASET}_star_info.csv', index_col='planet_id')\n        ids = df.index.astype(int)\n        # if cls.DATASET == 'train':\n        #     # keep a small subset for debugging like origina\n        #     return ids[30:35]\n        return ids","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-15T08:57:08.505677Z","iopub.execute_input":"2025-08-15T08:57:08.506268Z","iopub.status.idle":"2025-08-15T08:57:08.514755Z","shell.execute_reply.started":"2025-08-15T08:57:08.506235Z","shell.execute_reply":"2025-08-15T08:57:08.513646Z"}},"outputs":[],"execution_count":4},{"cell_type":"code","source":"class SignalProcessor:\n    def __init__(self, config):\n        self.cfg = config\n        self.adc_info = pd.read_csv(f\"{self.cfg.DATA_PATH}/adc_info.csv\")\n        self.planet_ids = Config.get_planet_ids()\n\n    def _apply_linear_corr(self, linear_corr, signal):\n        linear_corr_flipped = np.flip(linear_corr, axis=0)\n        corrected_signal = signal.copy()\n        \n        for x, y in itertools.product(range(signal.shape[1]), range(signal.shape[2])):\n            poly = np.poly1d(linear_corr_flipped[:, x, y])\n            corrected_signal[:, x, y] = poly(corrected_signal[:, x, y])\n            \n        return corrected_signal\n\n    def _calibrate_single_signal(self, planet_id, sensor, num_observation):\n        sensor_cfg = self.cfg.SENSOR_CONFIG[sensor]\n\n        signal = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_signal_{num_observation}.parquet\").to_numpy()\n        dark = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_{num_observation}/dark.parquet\").to_numpy()\n        dead = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_{num_observation}/dead.parquet\").to_numpy()\n        flat = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_{num_observation}/flat.parquet\").to_numpy()\n        linear_corr = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_{num_observation}/linear_corr.parquet\").values.astype(np.float64).reshape(sensor_cfg[\"linear_corr_shape\"])\n\n        signal = signal.reshape(sensor_cfg[\"raw_shape\"])\n        gain = self.adc_info[f\"{sensor}_adc_gain\"].iloc[0]\n        offset = self.adc_info[f\"{sensor}_adc_offset\"].iloc[0]\n        signal = signal / gain + offset\n\n        hot = sigma_clip(dark, sigma=5, maxiters=5).mask\n\n        if sensor == \"AIRS-CH0\":\n            signal = signal[:, :, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            linear_corr = linear_corr[:, :, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            dark = dark[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            dead = dead[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            flat = flat[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            hot = hot[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n        \n        base_dt, increment = sensor_cfg[\"dt_pattern\"]\n        dt = np.ones(len(signal)) * base_dt\n        dt[1::2] += increment\n        \n        signal = signal.clip(0)\n        signal = self._apply_linear_corr(linear_corr, signal)\n        signal -= dark * dt[:, np.newaxis, np.newaxis]\n        \n        flat = flat.reshape(sensor_cfg[\"calibrated_shape\"])\n        flat[dead.reshape(sensor_cfg[\"calibrated_shape\"])] = np.nan\n        flat[hot.reshape(sensor_cfg[\"calibrated_shape\"])] = np.nan\n        \n        signal = signal / flat\n        \n        return signal\n\n    def _preprocess_calibrated_signal(self, calibrated_signal, sensor):\n        sensor_cfg = self.cfg.SENSOR_CONFIG[sensor]\n        binning = sensor_cfg[\"binning\"]\n\n        if sensor == \"AIRS-CH0\":\n            signal_roi = calibrated_signal[:, 10:22, :]\n        elif sensor == \"FGS1\":\n            signal_roi = calibrated_signal[:, 10:22, 10:22]\n            signal_roi = signal_roi.reshape(signal_roi.shape[0], -1)\n        \n        mean_signal = np.nanmean(signal_roi, axis=1)\n\n        cds_signal = mean_signal[1::2] - mean_signal[0::2]\n\n        n_bins = cds_signal.shape[0] // binning\n        binned = np.array([\n            cds_signal[j*binning : (j+1)*binning].mean(axis=0) \n            for j in range(n_bins)\n        ])\n\n        if sensor == \"FGS1\":\n            binned = binned.reshape((binned.shape[0], 1))\n            \n        return binned\n\n    def _process_planet_sensor(self, args):\n        planet_id, sensor = args['planet_id'], args['sensor']\n        preprocessed = []\n        for num_observation in [0, 1]:\n            file_path = f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_signal_{num_observation}.parquet\"\n            if os.path.isfile(file_path):\n                # print(num_observation)\n                calibrated = self._calibrate_single_signal(planet_id, sensor, num_observation)\n                preprocessed.append(self._preprocess_calibrated_signal(calibrated, sensor))\n\n        preprocessed = np.array(preprocessed)  # shape = (num_obs, num_features)\n        # print(\"Shape:\", preprocessed.shape, \"Mean shape:\", preprocessed.mean(axis=0).shape)\n    \n        # Mean across observations (axis=0 if each row is one observation)\n        return preprocessed.mean(axis=0)\n\n    def process_all_data(self):\n        args_fgs1 = [dict(planet_id=planet_id, sensor=\"FGS1\") for planet_id in self.planet_ids]\n        preprocessed_fgs1 = pqdm(args_fgs1, self._process_planet_sensor, n_jobs=self.cfg.N_JOBS)\n\n        args_airs_ch0 = [dict(planet_id=planet_id, sensor=\"AIRS-CH0\") for planet_id in self.planet_ids]\n        preprocessed_airs_ch0 = pqdm(args_airs_ch0, self._process_planet_sensor, n_jobs=self.cfg.N_JOBS)\n\n        preprocessed_signal = np.concatenate(\n            [np.stack(preprocessed_fgs1), np.stack(preprocessed_airs_ch0)], axis=2\n        )\n        return preprocessed_signal","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-15T08:57:08.515828Z","iopub.execute_input":"2025-08-15T08:57:08.516188Z","iopub.status.idle":"2025-08-15T08:57:08.545233Z","shell.execute_reply.started":"2025-08-15T08:57:08.516148Z","shell.execute_reply":"2025-08-15T08:57:08.544264Z"}},"outputs":[],"execution_count":5},{"cell_type":"code","source":"cfg = Config()\nprint('Loading and preprocessing signals...')\nsp = SignalProcessor(cfg)\npreprocessed = sp.process_all_data()\nprint('Preprocessed shape:', preprocessed.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-15T08:57:08.546188Z","iopub.execute_input":"2025-08-15T08:57:08.546579Z","execution_failed":"2025-08-15T08:58:12.076Z"}},"outputs":[{"name":"stdout","text":"Loading and preprocessing signals...\n","output_type":"stream"},{"output_type":"display_data","data":{"text/plain":"QUEUEING TASKS | :   0%|          | 0/1100 [00:00<?, ?it/s]","application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"e11212d92eda48159bbbc6f9e4be4ef8"}},"metadata":{}},{"output_type":"display_data","data":{"text/plain":"PROCESSING TASKS | :   0%|          | 0/1100 [00:00<?, ?it/s]","application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"7e6423d5ce1149769617536396bda796"}},"metadata":{}}],"execution_count":null},{"cell_type":"code","source":"data_agg_2d = preprocessed.reshape(-1, 187 * 283)\n\ndf_summary = pd.DataFrame(data_agg_2d, index=Config.get_planet_ids())\ndf_summary.index.name = 'planet_id'\ndf_summary","metadata":{"trusted":true,"execution":{"execution_failed":"2025-08-15T08:58:12.077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df_summary.reshpae(-1, 187, 283)\ndf_summary.to_csv(\"preprocessed_signal.csv\")","metadata":{"trusted":true,"execution":{"execution_failed":"2025-08-15T08:58:12.078Z"}},"outputs":[],"execution_count":null}]}