{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70367,"databundleVersionId":9188054,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"3cffb3cb-d3a3-4338-be7b-e9a7a9754a0c","_cell_guid":"ad79b440-0d88-4e5a-954e-8ee1c09cb2d4","execution":{"iopub.status.busy":"2024-10-31T04:25:56.524545Z","iopub.execute_input":"2024-10-31T04:25:56.525587Z","iopub.status.idle":"2024-10-31T04:25:57.581265Z","shell.execute_reply.started":"2024-10-31T04:25:56.525523Z","shell.execute_reply":"2024-10-31T04:25:57.580144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas.api.types #GLL metric\nimport scipy.stats #GLL metric\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_squared_error\nimport pickle","metadata":{"execution":{"iopub.status.busy":"2024-10-31T04:25:57.583739Z","iopub.execute_input":"2024-10-31T04:25:57.584335Z","iopub.status.idle":"2024-10-31T04:25:59.476315Z","shell.execute_reply.started":"2024-10-31T04:25:57.584286Z","shell.execute_reply":"2024-10-31T04:25:59.475145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Metadata Files","metadata":{}},{"cell_type":"markdown","source":"**train_adc_info.csv**: Contains analog-to-digital (ADC) conversion parameters (gain and offset) for restoring the original dynamic range of the data. Also includes a star column identifying which star was used for that planet's simulation. star column = There are 2 Stars","metadata":{}},{"cell_type":"code","source":"pd.read_csv(\"/kaggle/input/ariel-data-challenge-2024/train_adc_info.csv\", )","metadata":{"_uuid":"1b84e11b-c22b-4575-a306-4efaf5100817","_cell_guid":"a896eb9f-e477-48f7-a6d3-62d4f8a66ab6","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-10-26T05:38:43.324576Z","iopub.execute_input":"2024-10-26T05:38:43.325503Z","iopub.status.idle":"2024-10-26T05:38:43.353381Z","shell.execute_reply.started":"2024-10-26T05:38:43.325457Z","shell.execute_reply":"2024-10-26T05:38:43.352191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**train_labels.csv**: Ground Truth Spectra. There are 673 planets with 283 spectra features","metadata":{}},{"cell_type":"code","source":"train_labels = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2024/train_labels.csv\")\ntrain_labels","metadata":{"_uuid":"0ff31ea6-e463-45c7-9a51-c2bfdaa2bd95","_cell_guid":"a0b2174f-6bec-4a36-a0c7-c7007e5c16ab","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-10-31T05:23:47.900827Z","iopub.execute_input":"2024-10-31T05:23:47.901275Z","iopub.status.idle":"2024-10-31T05:23:48.021112Z","shell.execute_reply.started":"2024-10-31T05:23:47.901236Z","shell.execute_reply":"2024-10-31T05:23:48.019509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**sample_submission.csv**: Predict a mean and uncertainty for each planet_id. Predicting 1 Planet with 566 columns. First 283 columns are Spectra and Next 283 columns are Uncertainty","metadata":{}},{"cell_type":"code","source":"pd.read_csv(\"/kaggle/input/ariel-data-challenge-2024/test_adc_info.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-10-26T05:38:53.462646Z","iopub.execute_input":"2024-10-26T05:38:53.463664Z","iopub.status.idle":"2024-10-26T05:38:53.480979Z","shell.execute_reply.started":"2024-10-26T05:38:53.463616Z","shell.execute_reply":"2024-10-26T05:38:53.479711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv(\"/kaggle/input/ariel-data-challenge-2024/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-10-26T05:40:21.598408Z","iopub.execute_input":"2024-10-26T05:40:21.598864Z","iopub.status.idle":"2024-10-26T05:40:21.643921Z","shell.execute_reply.started":"2024-10-26T05:40:21.598815Z","shell.execute_reply":"2024-10-26T05:40:21.642904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**axis_info.parquet**: Axis information for both instruments. Instruments = FGS1 and AIRS-CHO. h = Hour","metadata":{}},{"cell_type":"code","source":"pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2024/axis_info.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-10-26T05:40:25.264903Z","iopub.execute_input":"2024-10-26T05:40:25.265405Z","iopub.status.idle":"2024-10-26T05:40:25.306486Z","shell.execute_reply.started":"2024-10-26T05:40:25.265365Z","shell.execute_reply":"2024-10-26T05:40:25.305465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**wavelength.csv**: The wavelength grid for each ground truth spectrum in the dataset","metadata":{}},{"cell_type":"code","source":"wavelengths = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2024/wavelengths.csv\")\nwavelengths","metadata":{"execution":{"iopub.status.busy":"2024-10-29T07:16:44.962034Z","iopub.execute_input":"2024-10-29T07:16:44.962494Z","iopub.status.idle":"2024-10-29T07:16:44.997991Z","shell.execute_reply.started":"2024-10-29T07:16:44.962454Z","shell.execute_reply":"2024-10-29T07:16:44.996916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wavelengths_array = np.array(wavelengths)\nwavelengths_array = wavelengths_array.transpose()","metadata":{"execution":{"iopub.status.busy":"2024-10-29T07:16:55.659324Z","iopub.execute_input":"2024-10-29T07:16:55.659768Z","iopub.status.idle":"2024-10-29T07:16:55.665852Z","shell.execute_reply.started":"2024-10-29T07:16:55.659728Z","shell.execute_reply":"2024-10-29T07:16:55.664565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(wavelengths_array)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-29T07:17:00.536998Z","iopub.execute_input":"2024-10-29T07:17:00.537454Z","iopub.status.idle":"2024-10-29T07:17:00.857608Z","shell.execute_reply.started":"2024-10-29T07:17:00.537413Z","shell.execute_reply":"2024-10-29T07:17:00.856411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Signal Files","metadata":{}},{"cell_type":"markdown","source":"**FGS1**: The First Channel of Ariel's Fine Guidance System (FGS). The purpose is for centering, focusing, and guiding of the satellite but it will also provide high-precision photometry of the target star in the visible spectrum. It has a sensitivity between 0.60 and 0.80 µm","metadata":{}},{"cell_type":"markdown","source":"**AIRS-CH0**: The First Channel (CH0) of Ariel InfraRed Spectrometer (AIRS). an infrared spectrometer with a sensitivity between 1.95 and 3.90 µm, and has a resolving power of approximately R=100","metadata":{}},{"cell_type":"markdown","source":"The tables below are for planet_id = 2633183716","metadata":{}},{"cell_type":"code","source":"# /kaggle/input/ariel-data-challenge-2024/train/2633183716/FGS1_signal.parquet\n# /kaggle/input/ariel-data-challenge-2024/train/2633183716/AIRS-CH0_signal.parquet\n# /kaggle/input/ariel-data-challenge-2024/train/2633183716/AIRS-CH0_calibration/dead.parquet\n# /kaggle/input/ariel-data-challenge-2024/train/2633183716/AIRS-CH0_calibration/linear_corr.parquet\n# /kaggle/input/ariel-data-challenge-2024/train/2633183716/AIRS-CH0_calibration/read.parquet\n# /kaggle/input/ariel-data-challenge-2024/train/2633183716/AIRS-CH0_calibration/flat.parquet\n# /kaggle/input/ariel-data-challenge-2024/train/2633183716/AIRS-CH0_calibration/dark.parquet\n# /kaggle/input/ariel-data-challenge-2024/train/2633183716/FGS1_calibration/dead.parquet\n# /kaggle/input/ariel-data-challenge-2024/train/2633183716/FGS1_calibration/linear_corr.parquet\n# /kaggle/input/ariel-data-challenge-2024/train/2633183716/FGS1_calibration/read.parquet\n# /kaggle/input/ariel-data-challenge-2024/train/2633183716/FGS1_calibration/flat.parquet\n# /kaggle/input/ariel-data-challenge-2024/train/2633183716/FGS1_calibration/dark.parquet","metadata":{"_uuid":"177263d3-ea4b-4bd7-a8d0-2e952fb62c9e","_cell_guid":"3d95e8c7-44ce-4433-95c4-926defc3e421","jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-10-14T11:38:01.663439Z","iopub.execute_input":"2024-10-14T11:38:01.664217Z","iopub.status.idle":"2024-10-14T11:38:01.786606Z","shell.execute_reply.started":"2024-10-14T11:38:01.664173Z","shell.execute_reply":"2024-10-14T11:38:01.784839Z"},"collapsed":false},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**AIRS-CH0_signal.parquet**: Rows = 11250 Frames (Images). Columns = 11392 (32 x 356 Image) pixels. Each image is a constant time step, i.e., the images together is a time series sequence. To get the original pixel values, multiply the values with the AIRS-CH0_adc_gain values from train_adc_info.csv then add the AIRS-CH0_adc_offset values, do this with the correct planet_id","metadata":{}},{"cell_type":"code","source":"AIRS_CH0_signal = pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2024/train/2633183716/AIRS-CH0_signal.parquet\")\nAIRS_CH0_signal","metadata":{"execution":{"iopub.status.busy":"2024-10-29T08:23:15.344479Z","iopub.execute_input":"2024-10-29T08:23:15.344935Z","iopub.status.idle":"2024-10-29T08:23:18.25797Z","shell.execute_reply.started":"2024-10-29T08:23:15.344893Z","shell.execute_reply":"2024-10-29T08:23:18.256608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AIRS_CH0_signal.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-26T05:33:10.378487Z","iopub.execute_input":"2024-10-26T05:33:10.37891Z","iopub.status.idle":"2024-10-26T05:33:10.52567Z","shell.execute_reply.started":"2024-10-26T05:33:10.378871Z","shell.execute_reply":"2024-10-26T05:33:10.524565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AIRS_CH0_signal.describe()","metadata":{"execution":{"iopub.status.busy":"2024-10-26T05:32:28.029471Z","iopub.execute_input":"2024-10-26T05:32:28.030377Z","iopub.status.idle":"2024-10-26T05:32:51.162474Z","shell.execute_reply.started":"2024-10-26T05:32:28.030316Z","shell.execute_reply":"2024-10-26T05:32:51.161367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AIRS_CH0_array = np.array(AIRS_CH0_signal.iloc[0])\nAIRS_CH0_array = AIRS_CH0_array.reshape(32, 356)\n\nplt.imshow(AIRS_CH0_array)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T11:40:39.150177Z","iopub.execute_input":"2024-10-27T11:40:39.150587Z","iopub.status.idle":"2024-10-27T11:40:39.354684Z","shell.execute_reply.started":"2024-10-27T11:40:39.15055Z","shell.execute_reply":"2024-10-27T11:40:39.353771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**FGS1_signal.parquet**: Rows = 135000 Frames (Images). Columns = 1024 (32 x 32 Image) pixels. Each image is a 0.1 second time step, i.e., the images together is a time series sequence. To get the original pixel values, multiply the values with the FGS1_adc_gain values from train_adc_info.csv then add the FGS1_adc_offset values, do this with the correct planet_id","metadata":{}},{"cell_type":"code","source":"FGS1_signal = pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2024/train/2633183716/FGS1_signal.parquet\")\nFGS1_signal","metadata":{"_uuid":"e2fed243-fd26-4674-80be-a660be2742e6","_cell_guid":"5ad7f66f-7c24-4fb2-89f5-047b545d4e2b","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-10-30T08:38:08.117272Z","iopub.execute_input":"2024-10-30T08:38:08.117672Z","iopub.status.idle":"2024-10-30T08:38:10.135276Z","shell.execute_reply.started":"2024-10-30T08:38:08.117632Z","shell.execute_reply":"2024-10-30T08:38:10.133408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FGS1_signal.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-26T05:34:58.215125Z","iopub.execute_input":"2024-10-26T05:34:58.215985Z","iopub.status.idle":"2024-10-26T05:34:58.272617Z","shell.execute_reply.started":"2024-10-26T05:34:58.215939Z","shell.execute_reply":"2024-10-26T05:34:58.271644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FGS1_signal.describe()","metadata":{"execution":{"iopub.status.busy":"2024-10-26T05:35:07.254185Z","iopub.execute_input":"2024-10-26T05:35:07.255235Z","iopub.status.idle":"2024-10-26T05:35:12.302779Z","shell.execute_reply.started":"2024-10-26T05:35:07.255178Z","shell.execute_reply":"2024-10-26T05:35:12.301759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FGS1_array = np.array(FGS1_signal.iloc[0])\nFGS1_array = FGS1_array.reshape(32, 32)\n\nplt.imshow(FGS1_array)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T11:40:46.1687Z","iopub.execute_input":"2024-10-27T11:40:46.169143Z","iopub.status.idle":"2024-10-27T11:40:46.345626Z","shell.execute_reply.started":"2024-10-27T11:40:46.169102Z","shell.execute_reply":"2024-10-27T11:40:46.344516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfinal_image = np.zeros((32,32))\nfor i in range(len(FGS1_signal)):\n    array = np.array(FGS1_signal.iloc[i])\n    array = array.reshape(32, 32)\n    final_image += array","metadata":{"execution":{"iopub.status.busy":"2024-10-27T11:40:52.584668Z","iopub.execute_input":"2024-10-27T11:40:52.585605Z","iopub.status.idle":"2024-10-27T11:41:02.248524Z","shell.execute_reply.started":"2024-10-27T11:40:52.58556Z","shell.execute_reply":"2024-10-27T11:41:02.247452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(final_image)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T11:41:08.453163Z","iopub.execute_input":"2024-10-27T11:41:08.453645Z","iopub.status.idle":"2024-10-27T11:41:08.655502Z","shell.execute_reply.started":"2024-10-27T11:41:08.453546Z","shell.execute_reply":"2024-10-27T11:41:08.654155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_image","metadata":{"execution":{"iopub.status.busy":"2024-10-26T06:01:08.880495Z","iopub.execute_input":"2024-10-26T06:01:08.881238Z","iopub.status.idle":"2024-10-26T06:01:08.88992Z","shell.execute_reply.started":"2024-10-26T06:01:08.881191Z","shell.execute_reply":"2024-10-26T06:01:08.888759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calibration Files","metadata":{}},{"cell_type":"markdown","source":"Calibration files record the electronic characteristics of the sensor and serve as \"supporting frames\" used in image post-processing to image signal to noise ratio. All the files are either of size 32x356 (AIRS-CH0) or 32x32 (FGS1)","metadata":{}},{"cell_type":"markdown","source":"**dead.parquet**: Identifies dead or hot pixels on the sensor. Dead pixels do not respond to light, while hot pixels consistently produce high signal levels regardless of incoming light","metadata":{}},{"cell_type":"code","source":"pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2024/train/2633183716/AIRS-CH0_calibration/dead.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-10-22T12:17:36.941517Z","iopub.execute_input":"2024-10-22T12:17:36.942651Z","iopub.status.idle":"2024-10-22T12:17:37.014044Z","shell.execute_reply.started":"2024-10-22T12:17:36.9426Z","shell.execute_reply":"2024-10-22T12:17:37.012844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2024/train/2633183716/FGS1_calibration/dead.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:33:50.238295Z","iopub.execute_input":"2024-10-24T11:33:50.238746Z","iopub.status.idle":"2024-10-24T11:33:50.29623Z","shell.execute_reply.started":"2024-10-24T11:33:50.238686Z","shell.execute_reply":"2024-10-24T11:33:50.294856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**linear_corr.parquet**: Information about the linearity correction of the sensor. The response of the pixels in the detector becomes less linear as they fill with electrons, approaching the point of saturation, where the pixel can no longer collect additional electrons and its response to light becomes flat. For an accurate estimate of the signal, the instrument's response as a function of the received charge is calibrated, and the correction is calculated using a polynomial of degree n. This polynomial allows for the conversion of the number of electrons collected/measured by the pixel into the number of electrons that the detector would have generated with a linear response.","metadata":{}},{"cell_type":"code","source":"pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2024/train/2633183716/AIRS-CH0_calibration/linear_corr.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:32:50.140021Z","iopub.execute_input":"2024-10-24T11:32:50.140475Z","iopub.status.idle":"2024-10-24T11:32:50.222079Z","shell.execute_reply.started":"2024-10-24T11:32:50.140426Z","shell.execute_reply":"2024-10-24T11:32:50.220594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2024/train/2633183716/FGS1_calibration/linear_corr.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-10-22T11:35:08.449813Z","iopub.execute_input":"2024-10-22T11:35:08.450766Z","iopub.status.idle":"2024-10-22T11:35:08.50405Z","shell.execute_reply.started":"2024-10-22T11:35:08.450719Z","shell.execute_reply":"2024-10-22T11:35:08.502937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**read.parquet**: Read noise frames capture the electronic noise introduced during the readout process of the sensor. This noise is present even when no light falls on the detector","metadata":{}},{"cell_type":"code","source":"pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2024/train/2633183716/AIRS-CH0_calibration/read.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:27:39.479133Z","iopub.execute_input":"2024-10-24T11:27:39.479577Z","iopub.status.idle":"2024-10-24T11:27:39.58048Z","shell.execute_reply.started":"2024-10-24T11:27:39.479527Z","shell.execute_reply":"2024-10-24T11:27:39.579085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2024/train/2633183716/FGS1_calibration/read.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-10-22T11:37:53.209966Z","iopub.execute_input":"2024-10-22T11:37:53.210946Z","iopub.status.idle":"2024-10-22T11:37:53.272734Z","shell.execute_reply.started":"2024-10-22T11:37:53.210897Z","shell.execute_reply":"2024-10-22T11:37:53.271713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**flat.parquet**: Flat field frames are created by imaging a uniformly illuminated surface, i.e., like a control in an experiment, all the pixels are the \"same value\". They are used to correct for variations in pixel-to-pixel sensitivity and optical system irregularities","metadata":{}},{"cell_type":"code","source":"pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2024/train/2633183716/AIRS-CH0_calibration/flat.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:29:39.852876Z","iopub.execute_input":"2024-10-24T11:29:39.853343Z","iopub.status.idle":"2024-10-24T11:29:39.943023Z","shell.execute_reply.started":"2024-10-24T11:29:39.853299Z","shell.execute_reply":"2024-10-24T11:29:39.941676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2024/train/2633183716/FGS1_calibration/flat.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-10-22T11:37:23.110065Z","iopub.execute_input":"2024-10-22T11:37:23.110868Z","iopub.status.idle":"2024-10-22T11:37:23.173157Z","shell.execute_reply.started":"2024-10-22T11:37:23.110819Z","shell.execute_reply":"2024-10-22T11:37:23.172071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**dark.parquet**: Dark frames are exposures taken with the shutter closed, capturing the thermal noise and bias level of the sensor. These are used to subtract the dark current from science images","metadata":{}},{"cell_type":"code","source":"pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2024/train/2633183716/AIRS-CH0_calibration/dark.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-10-22T12:18:47.544468Z","iopub.execute_input":"2024-10-22T12:18:47.544906Z","iopub.status.idle":"2024-10-22T12:18:47.625151Z","shell.execute_reply.started":"2024-10-22T12:18:47.544861Z","shell.execute_reply":"2024-10-22T12:18:47.624037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2024/train/2633183716/FGS1_calibration/dark.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-10-22T12:18:55.476337Z","iopub.execute_input":"2024-10-22T12:18:55.477213Z","iopub.status.idle":"2024-10-22T12:18:55.534481Z","shell.execute_reply.started":"2024-10-22T12:18:55.477163Z","shell.execute_reply":"2024-10-22T12:18:55.533416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Combine all FGS1 files","metadata":{}},{"cell_type":"code","source":"%%time\n\nfinal_df = pd.DataFrame()\nfor planet_id in os.listdir(\"/kaggle/input/ariel-data-challenge-2024/train/\"):\n    fgs1_signal = pd.read_parquet(f\"/kaggle/input/ariel-data-challenge-2024/train/{planet_id}/FGS1_signal.parquet\")\n    \n    final_image = np.zeros((1,1024))\n    for i in range(len(fgs1_signal)):\n        array = np.array(fgs1_signal.iloc[i])\n        final_image += array\n    \n    final_row = pd.DataFrame(final_image/len(fgs1_signal)) # Calculating Mean\n    planet_id_col = pd.DataFrame(data=[int(planet_id)], columns=[\"planet_id\"])\n    final_row = pd.concat([planet_id_col, final_row], axis=1)\n    final_df = pd.concat([final_df, final_row])\n    \n    if len(final_df) == 10:\n        break\n\nfinal_df","metadata":{"execution":{"iopub.status.busy":"2024-10-31T05:24:15.367526Z","iopub.execute_input":"2024-10-31T05:24:15.36791Z","iopub.status.idle":"2024-10-31T05:25:43.245081Z","shell.execute_reply.started":"2024-10-31T05:24:15.367874Z","shell.execute_reply":"2024-10-31T05:25:43.243795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Machine Learning","metadata":{}},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"planet_id_list = final_df.planet_id.values\n\nX = final_df.drop(columns=[\"planet_id\"])\ny = train_labels.loc[train_labels[\"planet_id\"].isin(list(planet_id_list))].drop(columns=[\"planet_id\"])","metadata":{"execution":{"iopub.status.busy":"2024-10-31T05:26:38.464745Z","iopub.execute_input":"2024-10-31T05:26:38.466012Z","iopub.status.idle":"2024-10-31T05:26:38.479985Z","shell.execute_reply.started":"2024-10-31T05:26:38.465956Z","shell.execute_reply":"2024-10-31T05:26:38.478608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape, y.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-31T05:26:39.846638Z","iopub.execute_input":"2024-10-31T05:26:39.847013Z","iopub.status.idle":"2024-10-31T05:26:39.854864Z","shell.execute_reply.started":"2024-10-31T05:26:39.846971Z","shell.execute_reply":"2024-10-31T05:26:39.853528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_linear = LinearRegression()\nmodel_linear.fit(X,y)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T05:26:44.190954Z","iopub.execute_input":"2024-10-31T05:26:44.191326Z","iopub.status.idle":"2024-10-31T05:26:44.270272Z","shell.execute_reply.started":"2024-10-31T05:26:44.191291Z","shell.execute_reply":"2024-10-31T05:26:44.269002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pred = model_linear.predict(X)\ntrain_pred","metadata":{"execution":{"iopub.status.busy":"2024-10-31T05:26:47.275641Z","iopub.execute_input":"2024-10-31T05:26:47.276029Z","iopub.status.idle":"2024-10-31T05:26:47.302347Z","shell.execute_reply.started":"2024-10-31T05:26:47.275994Z","shell.execute_reply":"2024-10-31T05:26:47.301191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pred.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-31T05:26:52.341125Z","iopub.execute_input":"2024-10-31T05:26:52.341506Z","iopub.status.idle":"2024-10-31T05:26:52.349929Z","shell.execute_reply.started":"2024-10-31T05:26:52.341472Z","shell.execute_reply":"2024-10-31T05:26:52.34832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Uncertainty (Sigma)\nsigma_array = []\nfor i in range(len(train_pred)):\n    sigma = mean_squared_error(y.iloc[i], train_pred[i], squared=False)\n    sigma = np.full((283),sigma)\n    sigma_array.append(sigma)\n\nsigma_array = np.array(sigma_array)\nsigma_array.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-31T05:27:04.848322Z","iopub.execute_input":"2024-10-31T05:27:04.848833Z","iopub.status.idle":"2024-10-31T05:27:04.864684Z","shell.execute_reply.started":"2024-10-31T05:27:04.848794Z","shell.execute_reply":"2024-10-31T05:27:04.86334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_predictions = np.concatenate((train_pred,sigma_array),axis=1)\ntrain_predictions","metadata":{"execution":{"iopub.status.busy":"2024-10-31T05:27:07.625608Z","iopub.execute_input":"2024-10-31T05:27:07.626665Z","iopub.status.idle":"2024-10-31T05:27:07.636238Z","shell.execute_reply.started":"2024-10-31T05:27:07.626556Z","shell.execute_reply":"2024-10-31T05:27:07.634697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_predictions.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-31T05:27:10.687318Z","iopub.execute_input":"2024-10-31T05:27:10.68771Z","iopub.status.idle":"2024-10-31T05:27:10.695336Z","shell.execute_reply.started":"2024-10-31T05:27:10.687674Z","shell.execute_reply":"2024-10-31T05:27:10.693998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Gaussian Log Likelihood Metric (https://www.kaggle.com/code/metric/ariel-gaussian-log-likelihood)","metadata":{}},{"cell_type":"code","source":"class ParticipantVisibleError(Exception):\n    pass\n\n\ndef score(\n        solution: pd.DataFrame,\n        submission: pd.DataFrame,\n        #row_id_column_name: str,\n        naive_mean: float,\n        naive_sigma: float,\n        sigma_true: float\n    ) -> float:\n    '''\n    This is a Gaussian Log Likelihood based metric. For a submission, which contains the predicted mean (x_hat) and variance (x_hat_std),\n    we calculate the Gaussian Log-likelihood (GLL) value to the provided ground truth (x). We treat each pair of x_hat,\n    x_hat_std as a 1D gaussian, meaning there will be 283 1D gaussian distributions, hence 283 values for each test spectrum,\n    the GLL value for one spectrum is the sum of all of them.\n\n    Inputs:\n        - solution: Ground Truth spectra (from test set)\n            - shape: (nsamples, n_wavelengths)\n        - submission: Predicted spectra and errors (from participants)\n            - shape: (nsamples, n_wavelengths*2)\n        naive_mean: (float) mean from the train set.\n        naive_sigma: (float) standard deviation from the train set.\n        sigma_true: (float) essentially sets the scale of the outputs.\n    '''\n\n    #del solution[row_id_column_name]\n    #del submission[row_id_column_name]\n\n    if submission.min().min() < 0:\n        raise ParticipantVisibleError('Negative values in the submission')\n    for col in submission.columns:\n        if not pandas.api.types.is_numeric_dtype(submission[col]):\n            raise ParticipantVisibleError(f'Submission column {col} must be a number')\n\n    n_wavelengths = len(solution.columns)\n    if len(submission.columns) != n_wavelengths*2:\n        raise ParticipantVisibleError('Wrong number of columns in the submission')\n\n    y_pred = submission.iloc[:, :n_wavelengths].values\n    # Set a non-zero minimum sigma pred to prevent division by zero errors.\n    sigma_pred = np.clip(submission.iloc[:, n_wavelengths:].values, a_min=10**-15, a_max=None)\n    y_true = solution.values\n\n    GLL_pred = np.sum(scipy.stats.norm.logpdf(y_true, loc=y_pred, scale=sigma_pred))\n    GLL_true = np.sum(scipy.stats.norm.logpdf(y_true, loc=y_true, scale=sigma_true * np.ones_like(y_true)))\n    GLL_mean = np.sum(scipy.stats.norm.logpdf(y_true, loc=naive_mean * np.ones_like(y_true), scale=naive_sigma * np.ones_like(y_true)))\n\n    submit_score = (GLL_pred - GLL_mean)/(GLL_true - GLL_mean)\n    return float(submit_score) #np.clip(submit_score, 0.0, 1.0))","metadata":{"execution":{"iopub.status.busy":"2024-10-31T05:27:13.883504Z","iopub.execute_input":"2024-10-31T05:27:13.883957Z","iopub.status.idle":"2024-10-31T05:27:13.896145Z","shell.execute_reply.started":"2024-10-31T05:27:13.883917Z","shell.execute_reply":"2024-10-31T05:27:13.894808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"solution_labels = train_labels.loc[train_labels[\"planet_id\"].isin(list(planet_id_list))].drop(columns=[\"planet_id\"])","metadata":{"execution":{"iopub.status.busy":"2024-10-31T05:27:15.239138Z","iopub.execute_input":"2024-10-31T05:27:15.240211Z","iopub.status.idle":"2024-10-31T05:27:15.25065Z","shell.execute_reply.started":"2024-10-31T05:27:15.24015Z","shell.execute_reply":"2024-10-31T05:27:15.248768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"solution_labels","metadata":{"execution":{"iopub.status.busy":"2024-10-30T08:42:33.655002Z","iopub.execute_input":"2024-10-30T08:42:33.656032Z","iopub.status.idle":"2024-10-30T08:42:33.690492Z","shell.execute_reply.started":"2024-10-30T08:42:33.655981Z","shell.execute_reply":"2024-10-30T08:42:33.689274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score(solution=solution_labels, submission=pd.DataFrame(train_predictions), #row_id_column_name=\"planet_id\",\n      naive_mean=train_labels.values.mean(), naive_sigma=train_labels.values.std(),\n      sigma_true=0.000145)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T05:27:19.877827Z","iopub.execute_input":"2024-10-31T05:27:19.87886Z","iopub.status.idle":"2024-10-31T05:27:19.912214Z","shell.execute_reply.started":"2024-10-31T05:27:19.878795Z","shell.execute_reply":"2024-10-31T05:27:19.910769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"GLL = -3.613746653617852e+23","metadata":{}},{"cell_type":"markdown","source":"# Save the Model","metadata":{}},{"cell_type":"code","source":"filename = \"/kaggle/working/final_model.sav\"\n\nwith open(filename,mode='wb') as file:\n    pickle.dump(model_linear, file)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T06:01:50.17999Z","iopub.execute_input":"2024-10-31T06:01:50.180542Z","iopub.status.idle":"2024-10-31T06:01:50.190182Z","shell.execute_reply.started":"2024-10-31T06:01:50.1805Z","shell.execute_reply":"2024-10-31T06:01:50.188726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with open(filename, mode='rb') as file:\n#     model = pickle.load(file)\n#     print(\"model loaded\")","metadata":{"execution":{"iopub.status.busy":"2024-10-31T05:50:41.810169Z","iopub.execute_input":"2024-10-31T05:50:41.811183Z","iopub.status.idle":"2024-10-31T05:50:41.818305Z","shell.execute_reply.started":"2024-10-31T05:50:41.811143Z","shell.execute_reply":"2024-10-31T05:50:41.817089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}