{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70367,"databundleVersionId":9188054,"sourceType":"competition"}],"dockerImageVersionId":30746,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sea\nimport matplotlib.pyplot as plt\nimport os\npd.set_option('display.max_columns',500)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-08-03T16:58:01.321691Z","iopub.execute_input":"2024-08-03T16:58:01.322163Z","iopub.status.idle":"2024-08-03T16:58:01.328204Z","shell.execute_reply.started":"2024-08-03T16:58:01.32213Z","shell.execute_reply":"2024-08-03T16:58:01.326885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# There are 5 CSVs shared - quick look into each file ","metadata":{}},{"cell_type":"markdown","source":"### 1, 2) train/test_adc_info : \nContains analog-to-digital (ADC) conversion parameters (gain and offset) for restoring the original dynamic range of the data. Also includes a star column identifying which star was used for that planet's simulation.\n\n##### train file:\n- Has 673 rows and 6 columns\n- Primary Key: 'planet_id'\n- Has information about 'FGS1_adc' and 'AIRS-CH0_adc' offset and gain\n- Table contains information about 2 type of stars -- 0 and 1 (346 and 327 rows respectively)\n\n##### test file:\n- Has 1 row and 6 column","metadata":{}},{"cell_type":"code","source":"train_adc = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2024/train_adc_info.csv\")\ntest_adc = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2024/test_adc_info.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-08-03T15:46:00.636889Z","iopub.execute_input":"2024-08-03T15:46:00.637313Z","iopub.status.idle":"2024-08-03T15:46:00.658671Z","shell.execute_reply.started":"2024-08-03T15:46:00.637286Z","shell.execute_reply":"2024-08-03T15:46:00.657046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_adc.info()","metadata":{"execution":{"iopub.status.busy":"2024-08-03T15:33:38.739873Z","iopub.execute_input":"2024-08-03T15:33:38.740253Z","iopub.status.idle":"2024-08-03T15:33:38.759226Z","shell.execute_reply.started":"2024-08-03T15:33:38.740221Z","shell.execute_reply":"2024-08-03T15:33:38.758085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_adc.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-03T15:33:30.704646Z","iopub.execute_input":"2024-08-03T15:33:30.705143Z","iopub.status.idle":"2024-08-03T15:33:30.721731Z","shell.execute_reply.started":"2024-08-03T15:33:30.705106Z","shell.execute_reply":"2024-08-03T15:33:30.72051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Making sure 'planet_id' is the primary key\nassert(train_adc[\"planet_id\"].nunique() == train_adc.shape[0])","metadata":{"execution":{"iopub.status.busy":"2024-08-03T15:34:48.537909Z","iopub.execute_input":"2024-08-03T15:34:48.539027Z","iopub.status.idle":"2024-08-03T15:34:48.545383Z","shell.execute_reply.started":"2024-08-03T15:34:48.538979Z","shell.execute_reply":"2024-08-03T15:34:48.543889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_adc[\"star\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-03T15:39:39.689703Z","iopub.execute_input":"2024-08-03T15:39:39.690112Z","iopub.status.idle":"2024-08-03T15:39:39.70143Z","shell.execute_reply.started":"2024-08-03T15:39:39.69008Z","shell.execute_reply":"2024-08-03T15:39:39.700055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_adc.shape","metadata":{"execution":{"iopub.status.busy":"2024-08-03T15:46:31.073231Z","iopub.execute_input":"2024-08-03T15:46:31.073722Z","iopub.status.idle":"2024-08-03T15:46:31.082384Z","shell.execute_reply.started":"2024-08-03T15:46:31.073683Z","shell.execute_reply":"2024-08-03T15:46:31.080806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_adc","metadata":{"execution":{"iopub.status.busy":"2024-08-03T15:46:34.627024Z","iopub.execute_input":"2024-08-03T15:46:34.627427Z","iopub.status.idle":"2024-08-03T15:46:34.643865Z","shell.execute_reply.started":"2024-08-03T15:46:34.627395Z","shell.execute_reply":"2024-08-03T15:46:34.642208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3) train_labels.csv: \nGround truth spectra\n\n- Has 673 rows and 284 columns (planet_id, n_wavelengths)\n- Primary Key: 'planet_id'\n- The table has 283 wavelengths observed for the same planet. Goes from wl_1:wl_283\n- These are the same 673 'planet_id's seen in  'train_adc_info' file\n- The wavelength distribution is not consistent across planet Ids -- ranges from Left skewed to Right skewed\n- The star used for the planet's simulation can be identified from 'train_adc_info' file","metadata":{}},{"cell_type":"code","source":"train_labels = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2024/train_labels.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-08-03T15:50:16.493142Z","iopub.execute_input":"2024-08-03T15:50:16.494206Z","iopub.status.idle":"2024-08-03T15:50:16.596383Z","shell.execute_reply.started":"2024-08-03T15:50:16.49416Z","shell.execute_reply":"2024-08-03T15:50:16.594951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.shape","metadata":{"execution":{"iopub.status.busy":"2024-08-03T15:52:05.976629Z","iopub.execute_input":"2024-08-03T15:52:05.977436Z","iopub.status.idle":"2024-08-03T15:52:05.983994Z","shell.execute_reply.started":"2024-08-03T15:52:05.977397Z","shell.execute_reply":"2024-08-03T15:52:05.982742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-03T16:21:58.913261Z","iopub.execute_input":"2024-08-03T16:21:58.913772Z","iopub.status.idle":"2024-08-03T16:21:59.102019Z","shell.execute_reply.started":"2024-08-03T16:21:58.913726Z","shell.execute_reply":"2024-08-03T16:21:59.100892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Making sure 'planet_id' is the primary key\nassert(train_labels[\"planet_id\"].nunique() == train_labels.shape[0])","metadata":{"execution":{"iopub.status.busy":"2024-08-03T15:53:39.965532Z","iopub.execute_input":"2024-08-03T15:53:39.96668Z","iopub.status.idle":"2024-08-03T15:53:39.975427Z","shell.execute_reply.started":"2024-08-03T15:53:39.966627Z","shell.execute_reply":"2024-08-03T15:53:39.973832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The 673 planet_ids in both 'train_adc' and 'train_labels' file match\nsum(train_labels[\"planet_id\"] == train_adc[\"planet_id\"])","metadata":{"execution":{"iopub.status.busy":"2024-08-03T15:55:57.035512Z","iopub.execute_input":"2024-08-03T15:55:57.036047Z","iopub.status.idle":"2024-08-03T15:55:57.045233Z","shell.execute_reply.started":"2024-08-03T15:55:57.036Z","shell.execute_reply":"2024-08-03T15:55:57.043801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The table has 283 wavelengths observed for the same planet. Goes from wl_1:wl_283","metadata":{}},{"cell_type":"code","source":"list(train_labels.columns)[1:6],list(train_labels.columns)[-5:]","metadata":{"execution":{"iopub.status.busy":"2024-08-03T16:01:28.596853Z","iopub.execute_input":"2024-08-03T16:01:28.597279Z","iopub.status.idle":"2024-08-03T16:01:28.606901Z","shell.execute_reply.started":"2024-08-03T16:01:28.597248Z","shell.execute_reply":"2024-08-03T16:01:28.605225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This might not mean much since we need to evaluate the wavelength differently for each planet_ID\ntrain_labels.describe()","metadata":{"execution":{"iopub.status.busy":"2024-08-03T16:07:35.492799Z","iopub.execute_input":"2024-08-03T16:07:35.493244Z","iopub.status.idle":"2024-08-03T16:07:36.151067Z","shell.execute_reply.started":"2024-08-03T16:07:35.493209Z","shell.execute_reply":"2024-08-03T16:07:36.149597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The distribution of wavelength is extremely varied for the first 10 planet_ids","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(2,5, figsize=(15, 6))\nfig.subplots_adjust(hspace = .5, wspace=.001)\naxs = axs.ravel()\nfor i in range(10):\n#     plt.hist(train_labels.iloc[i,1:283])\n#     plt.show()\n    axs[i].hist(train_labels.iloc[i,1:283])\n    axs[i].set_title(f\"planet_ID: {train_labels.iloc[i,0]}\")","metadata":{"execution":{"iopub.status.busy":"2024-08-03T16:19:10.446798Z","iopub.execute_input":"2024-08-03T16:19:10.447911Z","iopub.status.idle":"2024-08-03T16:19:12.347932Z","shell.execute_reply.started":"2024-08-03T16:19:10.447863Z","shell.execute_reply":"2024-08-03T16:19:12.346879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4) wavelength.csv: \nThe wavelength grid for each ground truth spectrum in the dataset.","metadata":{}},{"cell_type":"code","source":"wavelengths_file = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2024/wavelengths.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-08-03T16:24:17.026225Z","iopub.execute_input":"2024-08-03T16:24:17.027736Z","iopub.status.idle":"2024-08-03T16:24:17.046341Z","shell.execute_reply.started":"2024-08-03T16:24:17.027696Z","shell.execute_reply":"2024-08-03T16:24:17.045219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wavelengths_file.shape","metadata":{"execution":{"iopub.status.busy":"2024-08-03T16:24:23.424271Z","iopub.execute_input":"2024-08-03T16:24:23.424718Z","iopub.status.idle":"2024-08-03T16:24:23.433348Z","shell.execute_reply.started":"2024-08-03T16:24:23.424684Z","shell.execute_reply":"2024-08-03T16:24:23.431313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wavelengths_file","metadata":{"execution":{"iopub.status.busy":"2024-08-03T16:24:31.348352Z","iopub.execute_input":"2024-08-03T16:24:31.348743Z","iopub.status.idle":"2024-08-03T16:24:31.508524Z","shell.execute_reply.started":"2024-08-03T16:24:31.348714Z","shell.execute_reply":"2024-08-03T16:24:31.506938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5) sample_submission.csv:\nSample Submission file -- While submitting the file. We need to ensure it has all columns needed from here\n\n- Has 1 row and 567 columns ( 1 planet_id + 566 columns)\n- 566 (283+283) - First 283 columns must be the spectra, and next 283 columns must be the uncertainties.","metadata":{"execution":{"iopub.status.busy":"2024-08-03T16:50:36.348891Z","iopub.execute_input":"2024-08-03T16:50:36.349301Z","iopub.status.idle":"2024-08-03T16:50:36.355737Z","shell.execute_reply.started":"2024-08-03T16:50:36.349268Z","shell.execute_reply":"2024-08-03T16:50:36.354448Z"}}},{"cell_type":"code","source":"sub = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2024/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-08-03T16:48:01.239652Z","iopub.execute_input":"2024-08-03T16:48:01.240059Z","iopub.status.idle":"2024-08-03T16:48:01.265976Z","shell.execute_reply.started":"2024-08-03T16:48:01.24003Z","shell.execute_reply":"2024-08-03T16:48:01.264698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.shape","metadata":{"execution":{"iopub.status.busy":"2024-08-03T16:48:01.953125Z","iopub.execute_input":"2024-08-03T16:48:01.953526Z","iopub.status.idle":"2024-08-03T16:48:01.961564Z","shell.execute_reply.started":"2024-08-03T16:48:01.95348Z","shell.execute_reply":"2024-08-03T16:48:01.960237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub","metadata":{"execution":{"iopub.status.busy":"2024-08-03T16:48:23.309926Z","iopub.execute_input":"2024-08-03T16:48:23.310987Z","iopub.status.idle":"2024-08-03T16:48:23.624813Z","shell.execute_reply.started":"2024-08-03T16:48:23.310943Z","shell.execute_reply":"2024-08-03T16:48:23.623484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# The 'train' directory Exploration\n\n- Train directory has 673 folders in it\n- There are two devices used in the data collection -- 'CH0', 'FGS1'\n- All the 'planet_id's from 'train_adc_info' file have a corresponding folder in train\n- Each 'planet_id' folder in train has 2 parquet files - 'AIRS-CH0_signal.parquet' and 'FGS1_signal.parquet' --  Each row is an image in these files\n- Each of the device ('CH0', 'FGS1') -- have 5 calliberation files namely -- \n    - dark.parquet\n    - dead.parquet\n    - flat.parquet\n    - linear_corr.parquet\n    - read.parquet\n","metadata":{}},{"cell_type":"code","source":"# Getting count of folders in train\npath = '/kaggle/input/ariel-data-challenge-2024/train'\nlen(os.listdir(path))","metadata":{"execution":{"iopub.status.busy":"2024-08-03T17:05:45.117406Z","iopub.execute_input":"2024-08-03T17:05:45.117855Z","iopub.status.idle":"2024-08-03T17:05:45.128473Z","shell.execute_reply.started":"2024-08-03T17:05:45.117822Z","shell.execute_reply":"2024-08-03T17:05:45.127013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check how many planet_ids data present in train folder\ntrain_files = os.listdir(path)\ntrain_adc_planets = list(train_adc[\"planet_id\"])\nmatch_count = 0\nfor i in train_files:\n    if int(i) in train_adc_planets:\n        match_count+=1\nprint(\"Total planet ids in 'train_adc' file: \",train_adc.shape[0])\nprint(\"Planet_id which has corresponding file in train directory:\", match_count)","metadata":{"execution":{"iopub.status.busy":"2024-08-03T17:07:26.222019Z","iopub.execute_input":"2024-08-03T17:07:26.22244Z","iopub.status.idle":"2024-08-03T17:07:26.236549Z","shell.execute_reply.started":"2024-08-03T17:07:26.222409Z","shell.execute_reply":"2024-08-03T17:07:26.234826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### AIRS-CH0_signal.parquet \n\n- Each file has a shape of 11250 rows and 11392 columns (11250 images each with pixel size of 32 * 356 (11392) ) -- numpy.reshape(11250, 32, 356)\n- To restore the full dynamic range you must multiply the data by the matching gain value from train_adc_info.csv and then add the offset value, also from train_adc_info.csv","metadata":{}},{"cell_type":"code","source":"planet_1_path = '/kaggle/input/ariel-data-challenge-2024/train/100468857'\nCH0_path = f'{planet_1_path}/AIRS-CH0_signal.parquet'\nCH0_file = pd.read_parquet(CH0_path)","metadata":{"execution":{"iopub.status.busy":"2024-08-03T17:55:59.67417Z","iopub.execute_input":"2024-08-03T17:55:59.674711Z","iopub.status.idle":"2024-08-03T17:56:00.843712Z","shell.execute_reply.started":"2024-08-03T17:55:59.674666Z","shell.execute_reply":"2024-08-03T17:56:00.8425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CH0_file.shape","metadata":{"execution":{"iopub.status.busy":"2024-08-03T17:56:08.1253Z","iopub.execute_input":"2024-08-03T17:56:08.125805Z","iopub.status.idle":"2024-08-03T17:56:08.134437Z","shell.execute_reply.started":"2024-08-03T17:56:08.125765Z","shell.execute_reply":"2024-08-03T17:56:08.133133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CH0_file.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-03T17:56:27.547779Z","iopub.execute_input":"2024-08-03T17:56:27.548175Z","iopub.status.idle":"2024-08-03T17:56:27.721746Z","shell.execute_reply.started":"2024-08-03T17:56:27.548145Z","shell.execute_reply":"2024-08-03T17:56:27.720229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### FGS1_signal.parquet\n\n- Each file has a shape of 135000 rows and 1024 columns (1024 images each with pixel size of 32 * 32 (1024) ) -- numpy.reshape(135000, 32, 32)\n\n\n\n- To restore the full dynamic range you must multiply the data by the matching gain value from train_adc_info.csv and then add the offset value, also from train_adc_info.csv","metadata":{"execution":{"iopub.status.busy":"2024-08-03T17:59:08.521363Z","iopub.execute_input":"2024-08-03T17:59:08.521842Z","iopub.status.idle":"2024-08-03T17:59:08.527785Z","shell.execute_reply.started":"2024-08-03T17:59:08.52181Z","shell.execute_reply":"2024-08-03T17:59:08.52613Z"}}},{"cell_type":"code","source":"FGS1_path = f'{planet_1_path}/FGS1_signal.parquet'\nFGS1_file = pd.read_parquet(FGS1_path)","metadata":{"execution":{"iopub.status.busy":"2024-08-03T17:59:43.089742Z","iopub.execute_input":"2024-08-03T17:59:43.09015Z","iopub.status.idle":"2024-08-03T17:59:43.641853Z","shell.execute_reply.started":"2024-08-03T17:59:43.090122Z","shell.execute_reply":"2024-08-03T17:59:43.640466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FGS1_file.shape","metadata":{"execution":{"iopub.status.busy":"2024-08-03T17:59:48.888875Z","iopub.execute_input":"2024-08-03T17:59:48.889252Z","iopub.status.idle":"2024-08-03T17:59:48.896383Z","shell.execute_reply.started":"2024-08-03T17:59:48.889223Z","shell.execute_reply":"2024-08-03T17:59:48.895324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FGS1_file.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-03T18:01:02.053785Z","iopub.execute_input":"2024-08-03T18:01:02.054873Z","iopub.status.idle":"2024-08-03T18:01:02.221739Z","shell.execute_reply.started":"2024-08-03T18:01:02.054822Z","shell.execute_reply":"2024-08-03T18:01:02.220579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### CHO and FGS1 - Calliberation files \n\n- dark.parquet-- Shape: 32*356 for cho and 32*32 for FGS1 -  Dark frames are exposures taken with the shutter closed, capturing the thermal noise and bias level of the sensor. These are used to subtract the dark current from science images. \n- dead.parquet--  Shape: 32*356 for cho and 32*32 for FGS1 -  Identifies dead or hot pixels on the sensor. Dead pixels do not respond to light, while hot pixels consistently produce high signal levels regardless of incoming light.\n- flat.parquet--  Shape: 32*356 for cho and 32*32 for FGS1 -  Flat field frames are created by imaging a uniformly illuminated surface. They are used to correct for variations in pixel-to-pixel sensitivity and optical system irregularities.\n- linear_corr.parquet-- Shape: 192*356 for cho and 192*32 for FGS1 -  Yet to explore and see how to use this\n- read.parquet--  Shape: 32*356 for cho and 32*32 for FGS1 -  Read noise frames capture the electronic noise introduced during the readout process of the sensor. This noise is present even when no light falls on the detector.","metadata":{"execution":{"iopub.status.busy":"2024-08-03T18:08:12.090217Z","iopub.execute_input":"2024-08-03T18:08:12.090667Z","iopub.status.idle":"2024-08-03T18:08:12.097414Z","shell.execute_reply.started":"2024-08-03T18:08:12.090635Z","shell.execute_reply":"2024-08-03T18:08:12.095761Z"}}},{"cell_type":"code","source":"planet_1_path = '/kaggle/input/ariel-data-challenge-2024/train/100468857'\ndevice_path = 'AIRS-CH0_calibration'\n\ncho_dark_frame = pd.read_parquet(f'{planet_1_path}/{device_path}/dark.parquet').values.reshape(1,32,356)\ncho_dead_frame = pd.read_parquet(f'{planet_1_path}/{device_path}/dead.parquet').values.reshape(1,32,356)\ncho_flat_frame = pd.read_parquet(f'{planet_1_path}/{device_path}/flat.parquet').values.reshape(1,32,356)\ncho_read_frame = pd.read_parquet(f'{planet_1_path}/{device_path}/read.parquet').values.reshape(1,32,356)\ncho_lin_cor_frame = pd.read_parquet(f'{planet_1_path}/{device_path}/linear_corr.parquet')","metadata":{"execution":{"iopub.status.busy":"2024-08-03T18:15:41.813691Z","iopub.execute_input":"2024-08-03T18:15:41.814621Z","iopub.status.idle":"2024-08-03T18:15:41.931686Z","shell.execute_reply.started":"2024-08-03T18:15:41.814569Z","shell.execute_reply":"2024-08-03T18:15:41.930464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"cho_dark_frame.shape\", cho_dark_frame.shape)\nprint(\"cho_dead_frame.shape\", cho_dead_frame.shape)  \nprint(\"cho_flat_frame.shape\", cho_flat_frame.shape)  \nprint(\"cho_read_frame.shape\", cho_read_frame.shape)  \nprint(\"cho_lin_cor_frame.shape\", cho_lin_cor_frame.shape)","metadata":{"execution":{"iopub.status.busy":"2024-08-03T18:18:17.425198Z","iopub.execute_input":"2024-08-03T18:18:17.425658Z","iopub.status.idle":"2024-08-03T18:18:17.432747Z","shell.execute_reply.started":"2024-08-03T18:18:17.425612Z","shell.execute_reply":"2024-08-03T18:18:17.43156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"planet_1_path = '/kaggle/input/ariel-data-challenge-2024/train/100468857'\ndevice_path = 'FGS1_calibration'\n\nFGS1_dark_frame = pd.read_parquet(f'{planet_1_path}/{device_path}/dark.parquet').values.reshape(1,32,32)\nFGS1_dead_frame = pd.read_parquet(f'{planet_1_path}/{device_path}/dead.parquet').values.reshape(1,32,32)\nFGS1_flat_frame = pd.read_parquet(f'{planet_1_path}/{device_path}/flat.parquet').values.reshape(1,32,32)\nFGS1_read_frame = pd.read_parquet(f'{planet_1_path}/{device_path}/read.parquet').values.reshape(1,32,32)\nFGS1_lin_cor_frame = pd.read_parquet(f'{planet_1_path}/{device_path}/linear_corr.parquet')","metadata":{"execution":{"iopub.status.busy":"2024-08-03T18:20:45.068436Z","iopub.execute_input":"2024-08-03T18:20:45.068988Z","iopub.status.idle":"2024-08-03T18:20:45.119698Z","shell.execute_reply.started":"2024-08-03T18:20:45.068951Z","shell.execute_reply":"2024-08-03T18:20:45.11857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"FGS1_dark_frame.shape\", FGS1_dark_frame.shape)\nprint(\"FGS1_dead_frame.shape\", FGS1_dead_frame.shape)  \nprint(\"FGS1_flat_frame.shape\", FGS1_flat_frame.shape)  \nprint(\"FGS1_read_frame.shape\", FGS1_read_frame.shape)  \nprint(\"FGS1_lin_cor_frame.shape\", FGS1_lin_cor_frame.shape)","metadata":{"execution":{"iopub.status.busy":"2024-08-03T18:22:28.21785Z","iopub.execute_input":"2024-08-03T18:22:28.218623Z","iopub.status.idle":"2024-08-03T18:22:28.2245Z","shell.execute_reply.started":"2024-08-03T18:22:28.218586Z","shell.execute_reply":"2024-08-03T18:22:28.223303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Function to Read (for given Planet_ID)\n- Image + Calliberation parquet files \n- Gain and offset \n- plot graph\n","metadata":{"execution":{"iopub.status.busy":"2024-08-03T18:30:06.221057Z","iopub.execute_input":"2024-08-03T18:30:06.22275Z","iopub.status.idle":"2024-08-03T18:30:06.23128Z","shell.execute_reply.started":"2024-08-03T18:30:06.222658Z","shell.execute_reply":"2024-08-03T18:30:06.230132Z"}}},{"cell_type":"code","source":"def read_train_data(planet_id,device):\n    device_id = 'AIRS-CH0' if device == 'CH0' else 'FGS1'\n    reshape_val = 356 if device == 'CH0' else 32\n    data_path = f'/kaggle/input/ariel-data-challenge-2024/train/{planet_id}'\n    img_file_loc = f'{data_path}/{device_id}_signal.parquet'\n    \n    img_file_data = pd.read_parquet(img_file_loc)\n    \n    calib_file_loc = f'{data_path}/{device_id}_calibration'\n    dark_frame = pd.read_parquet(f'{calib_file_loc}/dark.parquet').values.reshape(1,32,reshape_val)\n    read_frame = pd.read_parquet(f'{calib_file_loc}/read.parquet').values.reshape(1,32,reshape_val)\n    flat_frame = pd.read_parquet(f'{calib_file_loc}/flat.parquet').values.reshape(1,32,reshape_val)\n    dead_frame = pd.read_parquet(f'{calib_file_loc}/dead.parquet').values.reshape(1,32,reshape_val)\n    flat_frame[dead_frame] = 1\n    \n    return img_file_data, dark_frame, read_frame, flat_frame, dead_frame","metadata":{"execution":{"iopub.status.busy":"2024-08-03T18:55:09.518546Z","iopub.execute_input":"2024-08-03T18:55:09.519514Z","iopub.status.idle":"2024-08-03T18:55:09.527025Z","shell.execute_reply.started":"2024-08-03T18:55:09.519456Z","shell.execute_reply":"2024-08-03T18:55:09.525891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_gain_offset(train_adc,planet_id,device):\n    device_id = 'AIRS-CH0' if device == 'CH0' else 'FGS1'\n    \n    planet_correction = train_adc.loc[train_adc.planet_id==planet_id]\n    gain   = planet_correction[f'{device_id}_adc_gain'].item()    # scalar\n    offset = planet_correction[f'{device_id}_adc_offset'].item()  # scalar\n    \n    return gain,offset\n#img_file_data, dark_frame, read_frame, flat_frame, dead_frame = read_train_data(100468857,'CH0')\n#get_gain_offset(train_adc,100468857,'FGS1')","metadata":{"execution":{"iopub.status.busy":"2024-08-03T19:47:36.666391Z","iopub.execute_input":"2024-08-03T19:47:36.666849Z","iopub.status.idle":"2024-08-03T19:47:36.674105Z","shell.execute_reply.started":"2024-08-03T19:47:36.666812Z","shell.execute_reply":"2024-08-03T19:47:36.672594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_images(planet_id, device, train_adc, stack_size):\n    reshape_val = 356 if device == 'CH0' else 32\n    img_file_data, dark_frame, read_frame, flat_frame, dead_frame = read_train_data(planet_id,device)\n    gain,offset = get_gain_offset(train_adc,planet_id,device)\n    \n    img_stack = img_file_data.iloc[:stack_size, :].values.reshape(stack_size, 32, reshape_val)\n    \n    # Processing\n    img_stack = (img_stack - dark_frame - read_frame)/flat_frame\n    img_stack = img_stack*gain + offset\n    \n    return img_stack  \n# i_s = process_images(100468857,'CH0', train_adc, 10)","metadata":{"execution":{"iopub.status.busy":"2024-08-03T20:00:16.749608Z","iopub.execute_input":"2024-08-03T20:00:16.750088Z","iopub.status.idle":"2024-08-03T20:00:16.759213Z","shell.execute_reply.started":"2024-08-03T20:00:16.750049Z","shell.execute_reply":"2024-08-03T20:00:16.757942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# planet_id, device, train_adc, stack_size\ndef plot_img(planet_id, device,train_adc, n_row, n_col, f_size, stack_size):\n    image_stack = process_images(planet_id,device,train_adc, stack_size)\n    fig,axes = plt.subplots(n_row, n_col, figsize=f_size)\n    axes = axes.ravel()\n    for k in range(stack_size):\n        img_k = image_stack[k, :, :]\n        sea.heatmap(img_k, cmap=\"mako\", ax=axes[k])\n        axes[k].set_title(f\"CH0: Image Number {k+1} for planet\\n {planet_id}\", size=8)\n        axes[k].set_xticks([])\n        axes[k].set_yticks([])\n        axes[k].set_xticklabels([])\n    axes[k].set_yticklabels([])","metadata":{"execution":{"iopub.status.busy":"2024-08-03T20:04:52.765991Z","iopub.execute_input":"2024-08-03T20:04:52.766423Z","iopub.status.idle":"2024-08-03T20:04:52.774802Z","shell.execute_reply.started":"2024-08-03T20:04:52.766391Z","shell.execute_reply":"2024-08-03T20:04:52.773721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting images","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image_stack = process_images(100468857,'CH0', train_adc, 10)\nplot_img(100468857,\"CH0\",train_adc,2,5,(16,4),10)","metadata":{"execution":{"iopub.status.busy":"2024-08-03T20:04:55.116131Z","iopub.execute_input":"2024-08-03T20:04:55.116517Z","iopub.status.idle":"2024-08-03T20:05:00.710712Z","shell.execute_reply.started":"2024-08-03T20:04:55.116471Z","shell.execute_reply":"2024-08-03T20:05:00.709393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_img(100468857,'FGS1',train_adc,2,5,(16,4),10)","metadata":{"execution":{"iopub.status.busy":"2024-08-03T20:05:03.059751Z","iopub.execute_input":"2024-08-03T20:05:03.060697Z","iopub.status.idle":"2024-08-03T20:05:07.553636Z","shell.execute_reply.started":"2024-08-03T20:05:03.060658Z","shell.execute_reply":"2024-08-03T20:05:07.552578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Showing the image for the first 5 examples in both the devices","metadata":{}},{"cell_type":"code","source":"planet_id_list = [100468857, 1005054328,1011759019,1012051641,1012409820]\nfor pl in planet_id_list:\n    plot_img(pl,\"CH0\",train_adc,2,5,(16,4),10)","metadata":{"execution":{"iopub.status.busy":"2024-08-03T20:07:19.448003Z","iopub.execute_input":"2024-08-03T20:07:19.449313Z","iopub.status.idle":"2024-08-03T20:07:52.967722Z","shell.execute_reply.started":"2024-08-03T20:07:19.44927Z","shell.execute_reply":"2024-08-03T20:07:52.966572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"planet_id_list = [100468857, 1005054328,1011759019,1012051641,1012409820]\nfor pl in planet_id_list:\n    plot_img(pl,\"FGS1\",train_adc,2,5,(16,4),10)","metadata":{"execution":{"iopub.status.busy":"2024-08-03T20:09:18.095743Z","iopub.execute_input":"2024-08-03T20:09:18.096206Z","iopub.status.idle":"2024-08-03T20:09:46.232337Z","shell.execute_reply.started":"2024-08-03T20:09:18.09617Z","shell.execute_reply":"2024-08-03T20:09:46.23113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-08-03T17:43:26.402095Z","iopub.execute_input":"2024-08-03T17:43:26.402631Z","iopub.status.idle":"2024-08-03T17:43:26.411113Z","shell.execute_reply.started":"2024-08-03T17:43:26.402592Z","shell.execute_reply":"2024-08-03T17:43:26.409317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-08-03T17:43:29.939982Z","iopub.execute_input":"2024-08-03T17:43:29.941084Z","iopub.status.idle":"2024-08-03T17:43:29.948271Z","shell.execute_reply.started":"2024-08-03T17:43:29.941044Z","shell.execute_reply":"2024-08-03T17:43:29.947032Z"},"trusted":true},"execution_count":null,"outputs":[]}]}