{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70367,"databundleVersionId":9188054,"sourceType":"competition"}],"dockerImageVersionId":30746,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install fastparquet","metadata":{"execution":{"iopub.status.busy":"2024-08-18T19:03:27.69741Z","iopub.status.idle":"2024-08-18T19:03:27.69787Z","shell.execute_reply.started":"2024-08-18T19:03:27.697643Z","shell.execute_reply":"2024-08-18T19:03:27.697665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport multiprocessing\nimport fastparquet\nimport itertools\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pandas.api.types\nimport scipy.stats\nfrom functools import partial\nfrom tqdm import tqdm\n\n\n%matplotlib inline\n","metadata":{"execution":{"iopub.status.busy":"2024-08-18T19:03:27.699425Z","iopub.status.idle":"2024-08-18T19:03:27.699853Z","shell.execute_reply.started":"2024-08-18T19:03:27.69966Z","shell.execute_reply":"2024-08-18T19:03:27.699677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ARIEL Data Calibration + Dataset Creation (1st Iteration)\n\nThis notebook will preprocessing pipeline. This will include the calibration and some binning to create a dataset with the planet_id, signal data, star.","metadata":{}},{"cell_type":"code","source":"class ARIELDataLoader(): ## 8 Core multiprocessing -> 22 mins (I/O bound maybe)\n                         ## 4 Core multiprocessing -> 35 mins\n\n    def __init__(self, adc, labels, binning = 10, train=True,cut_inf = 39, cut_sup = 321,\n                 plot=True, processes = multiprocessing.cpu_count()):\n\n        self.adc        = pd.read_csv(adc)\n        self.labels     = pd.read_csv(labels)\n        self.cut_inf    = cut_inf\n        self.cut_sup    = cut_sup\n        self.binning    = binning\n        self.plot       = plot\n        self.train      = train\n        self.processes  = processes\n\n        self.planet_ids = list(self.adc[\"planet_id\"])\n\n    def __getitem__(self, index, signal):\n\n        if signal == \"FGS1_calibration\":\n            self.binning = 600\n            self.planet_id  = self.planet_ids[index]\n            planet_dir = os.path.join(\"/\", ROOT, \"train\" if self.train else \"test\", str(self.planet_id))\n            FGS1_C_dir = os.path.join(planet_dir, \"FGS1_calibration\")\n            AIRS_C_dir = os.path.join(planet_dir, \"AIRS-CH0_calibration\")\n            FGS1_gain, FGS1_offset = self.adc[self.adc[\"planet_id\"] == self.planet_id]['FGS1_adc_gain'], self.adc[self.adc[\"planet_id\"] == self.planet_id]['FGS1_adc_offset']\n            FGS1_data  = np.asarray(pd.read_parquet(os.path.join(planet_dir, 'FGS1_signal.parquet'), engine='fastparquet'), dtype =np.float32).reshape(135000, 32, 32)\n            FGS1_dark  = pd.read_parquet(os.path.join(FGS1_C_dir, 'dark.parquet'), engine='fastparquet').values.astype(np.float16).reshape((32, 32))\n            FGS1_dead  = pd.read_parquet(os.path.join(FGS1_C_dir, 'dead.parquet'), engine='fastparquet').values.astype(np.float16).reshape((32, 32))\n            FGS1_flat  = pd.read_parquet(os.path.join(FGS1_C_dir, 'flat.parquet'), engine='fastparquet').values.astype(np.float16).reshape((32, 32))\n           \n            AIRS_gain, AIRS_offset = self.adc[self.adc[\"planet_id\"] == self.planet_id]['AIRS-CH0_adc_gain'], self.adc[self.adc[\"planet_id\"] == self.planet_id]['AIRS-CH0_adc_offset']\n            AIRS_data  = pd.read_parquet(os.path.join(planet_dir, 'AIRS-CH0_signal.parquet'), engine='fastparquet')\n            AIRS_dark  = pd.read_parquet(os.path.join(AIRS_C_dir, 'dark.parquet'), engine='fastparquet').values.astype(np.float32).reshape((32, 356))[:, self.cut_inf:self.cut_sup]\n            AIRS_dead  = pd.read_parquet(os.path.join(AIRS_C_dir, 'dead.parquet'), engine='fastparquet').values.astype(np.float32).reshape((32, 356))[:, self.cut_inf:self.cut_sup]\n            AIRS_flat  = pd.read_parquet(os.path.join(AIRS_C_dir, 'flat.parquet'), engine='fastparquet').values.astype(np.float32).reshape((32, 356))[:, self.cut_inf:self.cut_sup]\n\n            FGS1_data = self.calibrate(FGS1_data, FGS1_gain, FGS1_offset, FGS1_dead, FGS1_flat, FGS1_dark)\n            FGS1_data = FGS1_data/FGS1_data.mean()\n\n            AIRS_data = np.asarray(AIRS_data, dtype =np.float32).reshape(11250, 32, 356)\n            AIRS_data = AIRS_data[:, :, self.cut_inf:self.cut_sup]\n\n            self.binning = 300\n            AIRS_data = self.calibrate(AIRS_data, AIRS_gain, AIRS_offset, AIRS_dead, AIRS_flat, AIRS_dark)\n            AIRS_data = AIRS_data/AIRS_data.mean()\n            FGS1_data = np.append(self.planet_id, FGS1_data)\n            return np.append(FGS1_data, AIRS_data)\n\n\n    def __iter__(self):\n        df_list = []\n        for i in [\"FGS1_calibration\"]:  ## Removing this causes issues\n            partial_getitem = partial(self.__getitem__, signal = i)\n            with multiprocessing.Pool(self.processes) as pool:\n                results = list(tqdm(pool.imap(partial_getitem, range(len(self.planet_ids))),\n                                    total=len(self.planet_ids),\n                                    desc=\"Loading Data\",\n                                    unit=\"planet\"))\n                results = np.vstack(results)\n                df = pd.DataFrame(results)\n                df_list.append(df)\n\n        df_list[0] = df_list[0].rename(columns= {0: \"planet_id\"})\n\n        joined_df = df_list[0]\n        stars = self.adc[[\"planet_id\", \"star\"]]\n        joined_df = joined_df.merge(stars, on = \"planet_id\")\n        joined_df = joined_df.merge(self.labels, on =\"planet_id\")\n\n        return joined_df\n\n\n\n    def calibrate(self, signal, gain, offset, dead, flat, dark):\n        adc_signal = self.ADC_convert(signal, gain, offset)\n        clean_signal = self.clean_flat_dark_dead(adc_signal, dead, flat, dark)\n        cds = self.get_cds(clean_signal)\n        binned_signal_time = self.bin_obs(cds, binning=self.binning).sum(axis = (1, 2))  # Summing over all except time (preserving transit depth information in a way)\n\n        return binned_signal_time\n\n    def ADC_convert(self, signal, gain, offset):\n        gain = np.asarray(gain).reshape(1, 1, -1)\n        offset = np.asarray(offset).reshape(1, 1, -1)\n        signal /= gain\n        signal += offset\n        return signal\n\n    def clean_flat_dark_dead(self, signal, dead, flat, dark):\n        if dead is not None and flat is not None and dark is not None:\n            flat_masked = np.ma.masked_where(dead, flat)\n            dark_masked = np.ma.masked_where(dead, dark)\n            flat_tiled = np.tile(flat_masked, (signal.shape[0], 1, 1))\n            dark_tiled = np.tile(dark_masked, (signal.shape[0], 1, 1))\n            dead_tiled = np.tile(dead, (signal.shape[0], 1, 1))\n            signal_masked = np.ma.masked_where(dead_tiled, signal)\n            corrected_signal = (signal_masked - dark_tiled) / (flat_tiled - dark_tiled)\n            return corrected_signal\n        else:\n            return signal\n\n    def get_cds(self, signal):\n        cds = signal[1::2, :, :] - signal[::2, :, :]\n        return cds\n\n    def bin_obs(self, cds_signal, binning):\n        cds_transposed = cds_signal.transpose(0, 2, 1)\n        cds_binned = np.zeros((cds_transposed.shape[0] // binning, cds_transposed.shape[1], cds_transposed.shape[2]))\n\n        for i in range(cds_transposed.shape[0] // binning):\n            cds_binned[i, :, :] = np.sum(cds_transposed[i*binning:(i+1)*binning, :, :], axis=0)\n\n        return cds_binned\n\n    def plot_spectra(self, ata):\n        data_plot = data \n        plt.plot(data_plot/data_plot.mean(), '-', alpha=0.3)\n        plt.title(\"Signal Plot\")\n        plt.show()\n\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-08-18T19:03:27.701299Z","iopub.status.idle":"2024-08-18T19:03:27.70172Z","shell.execute_reply.started":"2024-08-18T19:03:27.701484Z","shell.execute_reply":"2024-08-18T19:03:27.701498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT = os.path.join(\"kaggle\", \"input\", \"ariel-data-challenge-2024\") \nadc_file = os.path.join(\"/\", ROOT, \"train_adc_info.csv\")\nlabels_file = os.path.join(\"/\", ROOT, \"train_labels.csv\")\n\n\ndata_loader = ARIELDataLoader(adc=adc_file, labels=labels_file, train=True, plot = True, binning = 300)\n\n# Get data for the first planet (index 0)\ndata = data_loader.__iter__() \ndata.to_csv(\"Data.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-08-18T19:03:27.703262Z","iopub.status.idle":"2024-08-18T19:03:27.703814Z","shell.execute_reply.started":"2024-08-18T19:03:27.703508Z","shell.execute_reply":"2024-08-18T19:03:27.70353Z"},"trusted":true},"execution_count":null,"outputs":[]}]}