{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":70367,"databundleVersionId":9188054,"sourceType":"competition"},{"sourceId":201432108,"sourceType":"kernelVersion"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This script is designed to load a pre-trained model and use it to make predictions on new data. The model was trained to process exposures and produce a clean spectrum for each exoplanet, summarizing the rp/rs values across all wavelengths. The script will load the model, preprocess the input data, make predictions, and save the results for further analysis.\r\n\r\nThe primary objective of this challenge is to handle noisy exposures and extract meaningful spectra. The baseline approach involves using a 1D CNN to fit the mean value of the transmission spectra and a 2D CNN to retrieve atmospheric features. This script will help you apply the trained model to new data and evaluate its performance.\r\n\r\nFeel free to modify and extend this script to suit your specific needs and improve the prediction accuracy.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport itertools\nimport os\nimport glob\nfrom numpy.polynomial import Polynomial\nfrom tqdm import tqdm\nimport threading\nimport pickle\n\nfrom tensorflow.keras.models import load_model\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:04:52.246927Z","iopub.execute_input":"2024-10-17T09:04:52.248338Z","iopub.status.idle":"2024-10-17T09:04:52.254223Z","shell.execute_reply.started":"2024-10-17T09:04:52.248282Z","shell.execute_reply":"2024-10-17T09:04:52.253215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the paths\noutput_dir = '/kaggle/input/host-starter-solution'\n\npath_folder = '/kaggle/input/ariel-data-challenge-2024/' \n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-17T09:04:52.255745Z","iopub.execute_input":"2024-10-17T09:04:52.256058Z","iopub.status.idle":"2024-10-17T09:04:52.267749Z","shell.execute_reply.started":"2024-10-17T09:04:52.256026Z","shell.execute_reply":"2024-10-17T09:04:52.26684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calc_basic_stat(data, n_sigma_rejection=5, iter_num=5):\n    \"\"\"\n    Perform basic statistical calculations and reject outliers.\n    \n    Parameters:\n    data (array-like): The input data to be processed.\n    n_sigma_rejection (int): The number of standard deviations for outlier rejection.\n    iter_num (int): The number of iterations for outlier rejection.\n    \n    Returns:\n    array-like: A boolean mask indicating the outliers.\n    \"\"\"\n    stat = {}\n    indexValid = ~np.logical_or(np.isnan(data), np.isinf(data))\n    data_valid = data[indexValid]\n    stat['stdVal_Outlier'] = 0\n    stat['meanVal_Outlier'] = 0\n    th_max = 0\n    th_min = 0\n    for iter_cnt in range(iter_num):\n        stat['meanVal'] = np.mean(data_valid)\n        stat['stdVal'] = np.std(data_valid, ddof=1)\n        th_max = stat['meanVal'] + n_sigma_rejection * stat['stdVal']\n        th_min = stat['meanVal'] - n_sigma_rejection * stat['stdVal']\n        ind_to_reject = np.logical_or(data_valid > th_max, data_valid < th_min)\n        data_valid = data_valid[~ind_to_reject]\n\n    # Create a full mask for the original data\n    full_mask = np.zeros(data.shape, dtype=bool)\n    full_mask[indexValid] = np.isin(data[indexValid], data_valid, invert=True)\n\n    return full_mask\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:04:52.268834Z","iopub.execute_input":"2024-10-17T09:04:52.269133Z","iopub.status.idle":"2024-10-17T09:04:52.281388Z","shell.execute_reply.started":"2024-10-17T09:04:52.269102Z","shell.execute_reply":"2024-10-17T09:04:52.280438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ADC_convert(signal, gain, offset):\n    signal /= gain\n    signal += offset\n    return signal\n\ndef mask_hot_dead(signal, dead, dark):\n    hot = calc_basic_stat(\n        dark, n_sigma_rejection=5, iter_num=5)\n    hot = np.tile(hot, (signal.shape[0], 1, 1))\n    dead = np.tile(dead, (signal.shape[0], 1, 1))\n    signal = np.ma.masked_where(dead, signal, copy=False)\n    signal = np.ma.masked_where(hot, signal, copy=False)\n    return signal\n\ndef apply_linear_corr(linear_corr,clean_signal):\n    for x, y in itertools.product(\n                range(clean_signal.shape[1]), range(clean_signal.shape[2])\n            ):\n        poli = Polynomial(linear_corr[:, x, y])\n        clean_signal[:, x, y] = poli(clean_signal[:, x, y])\n    return clean_signal\n\ndef clean_dark(signal, dead, dark, dt):\n    dark = np.ma.masked_where(dead, dark)\n    dark = np.tile(dark, (signal.shape[0], 1, 1))\n    signal -= dark* dt[:, np.newaxis, np.newaxis]\n    return signal\n\ndef get_cds(signal, cds):\n    np.subtract(signal[:,1::2,:,:], signal[:,::2,:,:], out=cds)\n\ndef bin_obs(cds_signal,binning):\n    cds_transposed = cds_signal.transpose(0,1,3,2)\n    cds_binned = np.zeros((cds_transposed.shape[0], cds_transposed.shape[1]//binning, cds_transposed.shape[2], cds_transposed.shape[3]), dtype=np.float64)\n    for i in range(cds_transposed.shape[1]//binning):\n        cds_binned[:,i,:,:] = np.sum(cds_transposed[:,i*binning:(i+1)*binning,:,:], axis=1)\n    return cds_binned\n\ndef correct_flat_field(flat,dead, signal):\n    flat = flat.transpose(1, 0)\n    dead = dead.transpose(1, 0)\n    flat = np.ma.masked_where(dead, flat)\n    flat = np.tile(flat, (signal.shape[0], 1, 1))\n    signal = signal / flat\n    return signal\n\n## we will start by getting the index of the training data:\ndef get_index(files, CHUNKS_SIZE):\n    index = []\n    for file in files :\n        file_name = file.split('/')[-1]\n        if file_name.split('_')[0] == 'AIRS-CH0' and file_name.split('_')[1] == 'signal.parquet':\n            file_index = os.path.basename(os.path.dirname(file))\n            index.append(int(file_index))\n    index = np.array(index)\n    index = np.sort(index) \n    index = np.array_split(index, len(index)//CHUNKS_SIZE)\n    return index\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:04:52.28328Z","iopub.execute_input":"2024-10-17T09:04:52.283585Z","iopub.status.idle":"2024-10-17T09:04:52.301151Z","shell.execute_reply.started":"2024-10-17T09:04:52.283553Z","shell.execute_reply":"2024-10-17T09:04:52.300111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calibrate_data(nthreads, thread_idx):\n    AIRS_CH0_clean = np.zeros((CHUNKS_SIZE, 11250, 32, l), dtype=np.float64)\n    FGS1_clean = np.zeros((CHUNKS_SIZE, 135000, 32, 32), dtype=np.float64)\n    \n    AIRS_cds = np.zeros((AIRS_CH0_clean.shape[0], AIRS_CH0_clean.shape[1]//2, AIRS_CH0_clean.shape[2], AIRS_CH0_clean.shape[3]))\n    FGS1_cds = np.zeros((FGS1_clean.shape[0], FGS1_clean.shape[1]//2, FGS1_clean.shape[2], FGS1_clean.shape[3]))\n    \n    for n, index_chunk in tqdm(enumerate(index)):\n        if n % nthreads != thread_idx:\n            continue\n        for i in range (CHUNKS_SIZE):\n            df = pd.read_parquet(os.path.join(path_folder,f'test/{index_chunk[i]}/AIRS-CH0_signal.parquet'))\n            signal = df.values.astype(np.float64).reshape((df.shape[0], 32, 356))\n            gain = test_adc_info['AIRS-CH0_adc_gain'].loc[index_chunk[i]]\n            offset = test_adc_info['AIRS-CH0_adc_offset'].loc[index_chunk[i]]\n            signal = ADC_convert(signal, gain, offset)\n            dt_airs = axis_info['AIRS-CH0-integration_time'].dropna().values\n            chopped_signal = signal[:, :, cut_inf:cut_sup]\n            del df\n            # CLEANING THE DATA: AIRS\n            flat = pd.read_parquet(os.path.join(path_folder,f'test/{index_chunk[i]}/AIRS-CH0_calibration/flat.parquet')).values.astype(np.float64).reshape((32, 356))[:, cut_inf:cut_sup]\n            dark = pd.read_parquet(os.path.join(path_folder,f'test/{index_chunk[i]}/AIRS-CH0_calibration/dark.parquet')).values.astype(np.float64).reshape((32, 356))[:, cut_inf:cut_sup]\n            dead_airs = pd.read_parquet(os.path.join(path_folder,f'test/{index_chunk[i]}/AIRS-CH0_calibration/dead.parquet')).values.astype(np.float64).reshape((32, 356))[:, cut_inf:cut_sup]\n            linear_corr = pd.read_parquet(os.path.join(path_folder,f'test/{index_chunk[i]}/AIRS-CH0_calibration/linear_corr.parquet')).values.astype(np.float64).reshape((6, 32, 356))[:, :, cut_inf:cut_sup]\n            \n            chopped_signal = mask_hot_dead(chopped_signal, dead_airs, dark)\n            AIRS_CH0_clean[i] = chopped_signal\n\n            linear_corr_signal = apply_linear_corr(linear_corr,AIRS_CH0_clean[i])\n            AIRS_CH0_clean[i,:, :, :] = linear_corr_signal\n            del linear_corr\n            \n            cleaned_signal = clean_dark(AIRS_CH0_clean[i], dead_airs, dark, dt_airs)\n            AIRS_CH0_clean[i] = cleaned_signal\n            del dark\n\n            df = pd.read_parquet(os.path.join(path_folder,f'test/{index_chunk[i]}/FGS1_signal.parquet'))\n            fgs_signal = df.values.astype(np.float64).reshape((df.shape[0], 32, 32))\n\n            FGS1_gain = test_adc_info['FGS1_adc_gain'].loc[index_chunk[i]]\n            FGS1_offset = test_adc_info['FGS1_adc_offset'].loc[index_chunk[i]]\n\n            fgs_signal = ADC_convert(fgs_signal, FGS1_gain, FGS1_offset)\n            dt_fgs1 = np.ones(len(fgs_signal))*0.1\n            chopped_FGS1 = fgs_signal\n            del fgs_signal, df\n\n            # CLEANING THE DATA: FGS1\n            flat = pd.read_parquet(os.path.join(path_folder,f'test/{index_chunk[i]}/FGS1_calibration/flat.parquet')).values.astype(np.float64).reshape((32, 32))\n            dark = pd.read_parquet(os.path.join(path_folder,f'test/{index_chunk[i]}/FGS1_calibration/dark.parquet')).values.astype(np.float64).reshape((32, 32))\n            dead_fgs1 = pd.read_parquet(os.path.join(path_folder,f'test/{index_chunk[i]}/FGS1_calibration/dead.parquet')).values.astype(np.float64).reshape((32, 32))\n            linear_corr = pd.read_parquet(os.path.join(path_folder,f'test/{index_chunk[i]}/FGS1_calibration/linear_corr.parquet')).values.astype(np.float64).reshape((6, 32, 32))\n            \n            chopped_FGS1 = mask_hot_dead(chopped_FGS1, dead_fgs1, dark)\n            FGS1_clean[i] = chopped_FGS1\n            \n            linear_corr_signal = apply_linear_corr(linear_corr,FGS1_clean[i])\n            FGS1_clean[i,:, :, :] = linear_corr_signal\n            del linear_corr\n\n            cleaned_signal = clean_dark(FGS1_clean[i], dead_fgs1, dark,dt_fgs1)\n            FGS1_clean[i] = cleaned_signal\n            del dark\n\n        # SAVE DATA AND FREE SPACE\n        get_cds(AIRS_CH0_clean, AIRS_cds)\n        get_cds(FGS1_clean, FGS1_cds)\n\n        AIRS_CH0_clean.fill(0)\n        FGS1_clean.fill(0)\n\n        ## (Optional) Time Binning to reduce space\n        AIRS_cds_binned = bin_obs(AIRS_cds,binning=30)\n        FGS1_cds_binned = bin_obs(FGS1_cds,binning=30*12)\n\n        for i in range (CHUNKS_SIZE):\n            flat_airs = pd.read_parquet(os.path.join(path_folder,f'test/{index_chunk[i]}/AIRS-CH0_calibration/flat.parquet')).values.astype(np.float64).reshape((32, 356))[:, cut_inf:cut_sup]\n            flat_fgs = pd.read_parquet(os.path.join(path_folder,f'test/{index_chunk[i]}/FGS1_calibration/flat.parquet')).values.astype(np.float64).reshape((32, 32))\n            \n            corrected_AIRS_cds_binned = correct_flat_field(flat_airs,dead_airs, AIRS_cds_binned[i])\n            AIRS_cds_binned[i] = corrected_AIRS_cds_binned\n            corrected_FGS1_cds_binned = correct_flat_field(flat_fgs,dead_fgs1, FGS1_cds_binned[i])\n            FGS1_cds_binned[i] = corrected_FGS1_cds_binned\n            \n        ## save data\n        np.save(os.path.join(path_out, 'AIRS_clean_test_{}.npy'.format(n)), AIRS_cds_binned)\n        np.save(os.path.join(path_out, 'FGS1_test_{}.npy'.format(n)), FGS1_cds_binned)\n\nCHUNKS_SIZE = 1\nfiles = glob.glob(os.path.join(path_folder + 'test/', '*/*'))\n\nindex = get_index(files, CHUNKS_SIZE)\n\ntest_adc_info = pd.read_csv(os.path.join(path_folder, 'test_adc_info.csv'))\ntest_adc_info = test_adc_info.set_index('planet_id')\naxis_info = pd.read_parquet(os.path.join(path_folder, 'axis_info.parquet'))\n\ncut_inf, cut_sup = 39, 321\nl = cut_sup - cut_inf\n\nnthreads = 2\nthreads = []\n\npath_out = '/kaggle/tmp/data_light_raw/' # path to the folder to store the light data\nif not os.path.exists(path_out):\n    os.makedirs(path_out)\n    print(f\"Directory {path_out} created.\")\nelse:\n    print(f\"Directory {path_out} already exists.\")\n    \nfor i in range(nthreads):\n    threads.append(threading.Thread(target=calibrate_data, args=(nthreads, i)))\n    threads[i].start()\n\nfor t in threads:\n    t.join()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:04:52.460354Z","iopub.execute_input":"2024-10-17T09:04:52.460839Z","iopub.status.idle":"2024-10-17T09:05:19.205161Z","shell.execute_reply.started":"2024-10-17T09:04:52.460778Z","shell.execute_reply":"2024-10-17T09:05:19.204173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the variables to a file\nwith open(f'{output_dir}/output_var.pkl', 'rb') as f:\n    output_var = pickle.load(f)\n    \n# Print the values to check them\nfor key, value in output_var.items():\n    print(f\"{key}: {value}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:05:19.207676Z","iopub.execute_input":"2024-10-17T09:05:19.208106Z","iopub.status.idle":"2024-10-17T09:05:19.214758Z","shell.execute_reply.started":"2024-10-17T09:05:19.208058Z","shell.execute_reply":"2024-10-17T09:05:19.213773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_data (file, chunk_size, nb_files) : \n    data0 = np.load(file + '_0.npy', mmap_mode='r')\n    data_all = np.zeros((nb_files*chunk_size, data0.shape[1], data0.shape[2], data0.shape[3]), dtype=np.float64)\n    data_all[:chunk_size] = data0\n    for i in range (1, nb_files) : \n        data_all[i*chunk_size:(i+1)*chunk_size] = np.load(file + '_{}.npy'.format(i), mmap_mode='r')\n    return data_all \n\nsignal_AIRS_diff_transposed_binned = load_data(path_out + 'AIRS_clean_test', CHUNKS_SIZE, len(index)) \nsignal_FGS_diff_transposed_binned = load_data(path_out + 'FGS1_test', CHUNKS_SIZE, len(index))\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:05:19.216015Z","iopub.execute_input":"2024-10-17T09:05:19.216371Z","iopub.status.idle":"2024-10-17T09:05:19.256285Z","shell.execute_reply.started":"2024-10-17T09:05:19.216338Z","shell.execute_reply":"2024-10-17T09:05:19.255335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inferencee","metadata":{}},{"cell_type":"code","source":"do_the_mcdropout_wc = False\ndo_the_mcdropout = False\n\nnb_dropout = 1000\nnb_dropout_wc = 1000\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:05:19.257717Z","iopub.execute_input":"2024-10-17T09:05:19.258407Z","iopub.status.idle":"2024-10-17T09:05:19.263095Z","shell.execute_reply.started":"2024-10-17T09:05:19.25836Z","shell.execute_reply":"2024-10-17T09:05:19.262128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FGS_column = signal_FGS_diff_transposed_binned.sum(axis = 2)\ndataset = np.concatenate([signal_AIRS_diff_transposed_binned, FGS_column[:,:, np.newaxis,:]], axis = 2)\ndataset = dataset.sum(axis=3)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:05:19.266354Z","iopub.execute_input":"2024-10-17T09:05:19.267109Z","iopub.status.idle":"2024-10-17T09:05:19.279195Z","shell.execute_reply.started":"2024-10-17T09:05:19.267072Z","shell.execute_reply":"2024-10-17T09:05:19.278064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def norm_star_spectrum (signal) : \n    img_star = signal[:,:50].mean(axis = 1) + signal[:,-50:].mean(axis = 1)\n    return signal/img_star[:,np.newaxis,:]\n\ndataset_norm = norm_star_spectrum(dataset)\ndataset_norm = np.transpose(dataset_norm,(0,2,1))\n\nsignal_AIRS_diff_transposed_binned = signal_AIRS_diff_transposed_binned.sum(axis=3)\nwc_mean = signal_AIRS_diff_transposed_binned.mean(axis=1).mean(axis=1)\nwhite_curve = signal_AIRS_diff_transposed_binned.sum(axis=2)/ wc_mean[:, np.newaxis]\n\nprint(dataset_norm.shape)\nprint(white_curve.shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:05:19.280459Z","iopub.execute_input":"2024-10-17T09:05:19.28082Z","iopub.status.idle":"2024-10-17T09:05:19.29134Z","shell.execute_reply.started":"2024-10-17T09:05:19.280785Z","shell.execute_reply":"2024-10-17T09:05:19.290377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Normalize the wlc\nwlc_train_min = white_curve.min()\nwlc_train_max = white_curve.max()\nvalid_wc = (white_curve - wlc_train_min) / (wlc_train_max - wlc_train_min)\n# valid_wc = normalise_wlc(white_curve, output_var[\"wlc_train_min\"], output_var[\"wlc_train_max\"])","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:05:19.292773Z","iopub.execute_input":"2024-10-17T09:05:19.293059Z","iopub.status.idle":"2024-10-17T09:05:19.298146Z","shell.execute_reply.started":"2024-10-17T09:05:19.293028Z","shell.execute_reply":"2024-10-17T09:05:19.297095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1D CNN","metadata":{}},{"cell_type":"code","source":"model_path = f'{output_dir}/output/model_1dcnn.keras'\n# Load the model\nmodel_1d = load_model(model_path)\n\ndef unstandardizing (data, min_train_valid, max_train_valid) : \n    return data * (max_train_valid - min_train_valid) + min_train_valid\n\ndef MC_dropout_WC (model, data, nb_dropout) : \n    predictions = np.zeros((nb_dropout, data.shape[0]))\n    for i in range(nb_dropout) : \n        predictions[i,:] = model.predict(data, verbose = 0).flatten()\n    return predictions\n\nif do_the_mcdropout_wc :\n    print('Running ...')\n    prediction_valid_wc = MC_dropout_WC(model_1d, valid_wc, nb_dropout_wc)\n    spectre_valid_wc_all = unstandardizing(prediction_valid_wc, output_var['min_train_valid_wc'], output_var['max_train_valid_wc'])\n    spectre_valid_wc, spectre_valid_std_wc = spectre_valid_wc_all.mean(axis = 0), spectre_valid_wc_all.std(axis = 0)\n    print('Done.')\nelse : \n    spectre_valid_wc = model_1d.predict(valid_wc).flatten()\n    spectre_valid_wc = unstandardizing(spectre_valid_wc, output_var['min_train_valid_wc'], output_var['max_train_valid_wc'])\n    spectre_valid_std_wc = 0.1*np.abs(spectre_valid_wc)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:05:19.299566Z","iopub.execute_input":"2024-10-17T09:05:19.299958Z","iopub.status.idle":"2024-10-17T09:05:19.79704Z","shell.execute_reply.started":"2024-10-17T09:05:19.299915Z","shell.execute_reply":"2024-10-17T09:05:19.79619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(spectre_valid_wc)\nprint(spectre_valid_std_wc)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:05:19.798244Z","iopub.execute_input":"2024-10-17T09:05:19.79855Z","iopub.status.idle":"2024-10-17T09:05:19.804406Z","shell.execute_reply.started":"2024-10-17T09:05:19.798518Z","shell.execute_reply":"2024-10-17T09:05:19.803478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2D CNN","metadata":{}},{"cell_type":"code","source":"checkpoint_filepath = '/kaggle/input/host-starter-solution/output/model_2dcnn.keras'\nmodel_2d = load_model(checkpoint_filepath)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:05:19.805837Z","iopub.execute_input":"2024-10-17T09:05:19.806148Z","iopub.status.idle":"2024-10-17T09:05:24.263609Z","shell.execute_reply.started":"2024-10-17T09:05:19.806114Z","shell.execute_reply":"2024-10-17T09:05:24.262709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"###### Transpose #####\ndataset_norm = dataset_norm.transpose(0, 2, 1)\nprint(dataset_norm.shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:05:24.264886Z","iopub.execute_input":"2024-10-17T09:05:24.265206Z","iopub.status.idle":"2024-10-17T09:05:24.270572Z","shell.execute_reply.started":"2024-10-17T09:05:24.265172Z","shell.execute_reply":"2024-10-17T09:05:24.269627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##### Substracting the out transit signal #####\ndef suppress_out_transit (data, ingress, egress) : \n    data_in = data[:, ingress:egress,:]\n    return data_in\n\ningress, egress = 75,115\n\nvalid_obs_in = suppress_out_transit(dataset_norm, ingress, egress)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:05:24.271948Z","iopub.execute_input":"2024-10-17T09:05:24.272919Z","iopub.status.idle":"2024-10-17T09:05:24.280056Z","shell.execute_reply.started":"2024-10-17T09:05:24.272884Z","shell.execute_reply":"2024-10-17T09:05:24.279204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"###### Substract the mean #####\ndef substract_data_mean(data):\n    data_mean = np.zeros(data.shape)\n    for i in range(data.shape[0]):\n        data_mean[i] = data[i] - data[i].mean()\n    return data_mean\n\nvalid_obs_2d_mean = substract_data_mean(valid_obs_in)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:05:24.281269Z","iopub.execute_input":"2024-10-17T09:05:24.28159Z","iopub.status.idle":"2024-10-17T09:05:24.289546Z","shell.execute_reply.started":"2024-10-17T09:05:24.281558Z","shell.execute_reply":"2024-10-17T09:05:24.288608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##### Normalization dataset #####\ndef data_norm(data1):\n    data_min = data1.min()\n    data_max = data1.max()\n    data_abs_max = np.max([data_min, data_max])  \n    data1 = data1/data_abs_max\n    return data1, data_abs_max\n\nvalid_obs_norm, _ = data_norm(valid_obs_2d_mean)\n\nprint(valid_obs_norm.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:05:24.293335Z","iopub.execute_input":"2024-10-17T09:05:24.293808Z","iopub.status.idle":"2024-10-17T09:05:24.30086Z","shell.execute_reply.started":"2024-10-17T09:05:24.293732Z","shell.execute_reply":"2024-10-17T09:05:24.299907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def NN_uncertainity(model, x_test, targets_abs_max, T=5):\n    predictions = []\n    for _ in range(T):\n        pred_norm = model.predict([x_test],verbose=0)\n        pred = pred_norm * output_var['targets_abs_max']\n        predictions += [pred]  \n    mean, std = np.mean(np.array(predictions), axis=0), np.std(np.array(predictions), axis=0)\n    return mean, std\n\n\nif do_the_mcdropout :\n    spectre_valid_shift, spectre_valid_shift_std = NN_uncertainity(model_2d, [valid_obs_norm], output_var['targets_abs_max'], T = nb_dropout)\n    \nelse :\n    pred_valid_norm = model_2d.predict([valid_obs_norm])\n    pred_valid = pred_valid_norm * output_var['targets_abs_max']\n    spectre_valid_shift = pred_valid\n    spectre_valid_shift_std = spectre_valid_shift*0.1\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:05:24.30201Z","iopub.execute_input":"2024-10-17T09:05:24.302979Z","iopub.status.idle":"2024-10-17T09:05:24.714183Z","shell.execute_reply.started":"2024-10-17T09:05:24.30293Z","shell.execute_reply":"2024-10-17T09:05:24.713199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Combine 1D and 2D CNN output for FINAL SPECTRA","metadata":{}},{"cell_type":"code","source":"######## ADD THE FLUCTUATIONS TO THE MEAN ########\ndef add_the_mean (shift, mean) : \n    return shift + mean[:,np.newaxis]\n\npredictions_valid = add_the_mean(spectre_valid_shift, spectre_valid_wc)\npredictions_std_valid = np.sqrt(spectre_valid_std_wc[:,np.newaxis]**2 + spectre_valid_shift_std**2)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:05:24.715581Z","iopub.execute_input":"2024-10-17T09:05:24.716368Z","iopub.status.idle":"2024-10-17T09:05:24.72192Z","shell.execute_reply.started":"2024-10-17T09:05:24.716315Z","shell.execute_reply":"2024-10-17T09:05:24.720943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/sample_submission.csv')\n\npreds = predictions_valid.clip(0)\n# sigmas = predictions_std_valid\nsigmas = np.ones_like(preds) * (0.0001422*0.97 + 0.000176*0.03)\ndf_submission = pd.DataFrame(np.concatenate([preds,sigmas], axis=1), columns=ss.columns[1:])\ndf_submission.index = test_adc_info.index\n\n# Add a column at the beginning that is the same as the current index\ndf_submission.insert(0, 'planet_id', df_submission.index)\n# Reset the index to 0, 1, 2, ...\ndf_submission.reset_index(drop=True, inplace=True)\ndf_submission.to_csv('submission.csv', index=False, float_format='%.7f')\n\ndf_submission","metadata":{"execution":{"iopub.status.busy":"2024-10-17T09:05:24.723234Z","iopub.execute_input":"2024-10-17T09:05:24.723558Z","iopub.status.idle":"2024-10-17T09:05:24.773247Z","shell.execute_reply.started":"2024-10-17T09:05:24.72351Z","shell.execute_reply":"2024-10-17T09:05:24.772263Z"},"trusted":true},"execution_count":null,"outputs":[]}]}