{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70367,"databundleVersionId":9188054,"sourceType":"competition"},{"sourceId":9663942,"sourceType":"datasetVersion","datasetId":5904562}],"dockerImageVersionId":30746,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### ℹ️ **Info**\n* **forked original great work kernels**\n    * https://www.kaggle.com/code/sergeifironov/ariel-only-correlation\n\n* **2024/09/08 My Changed**\n    * scipy minimize() param & other params update\n* **2024/09/22 My Changed**\n    * improve LB .517 -> .522\n    \nhttps://www.kaggle.com/code/pourchot/ariel-data-challenge-2024　の改善を取り込む\n\n微分のphase_detectorの方式にする\ndeltaの工夫も取り入れる","metadata":{}},{"cell_type":"markdown","source":"---\n---","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport seaborn as sns\nimport scipy.stats\nfrom tqdm import tqdm\n\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.linear_model import Ridge\nfrom sklearn.metrics import r2_score, mean_squared_error\nimport itertools\nfrom scipy.optimize import minimize\nfrom functools import partial\nimport random, os\nfrom astropy.stats import sigma_clip\nimport warnings\nfrom sklearn.exceptions import ConvergenceWarning\nfrom sklearn.preprocessing import PolynomialFeatures\nfrom sklearn.linear_model import HuberRegressor, LinearRegression\nfrom multiprocessing import Pool\nimport pickle","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-19T06:19:22.69754Z","iopub.execute_input":"2024-10-19T06:19:22.698101Z","iopub.status.idle":"2024-10-19T06:19:22.706939Z","shell.execute_reply.started":"2024-10-19T06:19:22.698064Z","shell.execute_reply":"2024-10-19T06:19:22.70569Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"debug_num = int(1e9)\nmode = \"train\"\ntest_debug = False\nl = 30","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-19T06:43:12.744961Z","iopub.execute_input":"2024-10-19T06:43:12.745463Z","iopub.status.idle":"2024-10-19T06:43:12.752371Z","shell.execute_reply.started":"2024-10-19T06:43:12.745427Z","shell.execute_reply":"2024-10-19T06:43:12.75106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if mode == \"train\":\n    data_adc_info = pd.read_csv('/kaggle/input/neurips-ariel2024-train-fold/train_adc_info.csv',\n                               index_col='planet_id')\nelse:\n    data_adc_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/test_adc_info.csv',\n                               index_col='planet_id')\n\naxis_info = pd.read_parquet('/kaggle/input/ariel-data-challenge-2024/axis_info.parquet')","metadata":{"execution":{"iopub.status.busy":"2024-10-19T06:19:22.724222Z","iopub.execute_input":"2024-10-19T06:19:22.7247Z","iopub.status.idle":"2024-10-19T06:19:22.760621Z","shell.execute_reply.started":"2024-10-19T06:19:22.724656Z","shell.execute_reply":"2024-10-19T06:19:22.759548Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def apply_linear_corr(linear_corr,clean_signal):\n    linear_corr = np.flip(linear_corr, axis=0)\n    for x, y in itertools.product(\n                range(clean_signal.shape[1]), range(clean_signal.shape[2])\n            ):\n        poli = np.poly1d(linear_corr[:, x, y])\n        clean_signal[:, x, y] = poli(clean_signal[:, x, y])\n    return clean_signal\n\ndef clean_dark(signal, dark, dt):\n    dark = np.tile(dark, (signal.shape[0], 1, 1))\n    signal -= dark* dt[:, np.newaxis, np.newaxis]\n    return signal\n\n\ndef read_planet(z, dataset, sensor, binning):\n    planet_id, gain, offset = z\n    cut_inf, cut_sup = 39-l, 321+l\n    sensor_sizes_dict = {\"AIRS-CH0\":[[11250, 32, 356], [1, 32, cut_sup-cut_inf]], \"FGS1\":[[135000, 32, 32], [1, 32, 32]]}\n    binned_dict = {\"AIRS-CH0\":[11250 // binning // 2, 282+2*l], \"FGS1\":[135000 // binning // 2]}\n    linear_corr_dict = {\"AIRS-CH0\":(6, 32, 356), \"FGS1\":(6, 32, 32)}\n\n    signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/{dataset}/{planet_id}/{sensor}_signal.parquet').to_numpy()\n    dark_frame = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/{dataset}/' + str(planet_id) + '/' + sensor + '_calibration/dark.parquet', engine='pyarrow').to_numpy()\n    dead_frame = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/{dataset}/' + str(planet_id) + '/' + sensor + '_calibration/dead.parquet', engine='pyarrow').to_numpy()\n    flat_frame = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/{dataset}/' + str(planet_id) + '/' + sensor + '_calibration/flat.parquet', engine='pyarrow').to_numpy()\n    linear_corr = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/{dataset}/' + str(planet_id) + '/' + sensor + '_calibration/linear_corr.parquet').values.astype(np.float64).reshape(linear_corr_dict[sensor])\n\n    signal = signal.reshape(sensor_sizes_dict[sensor][0]) \n    # gain = adc_info[f'{sensor}_adc_gain'].values[i]\n    # offset = adc_info[f'{sensor}_adc_offset'].values[i]\n    signal = signal / gain + offset\n    \n    hot = sigma_clip(\n        dark_frame, sigma=8, maxiters=5\n    ).mask\n    \n    if sensor != \"FGS1\":\n        # 波長方向のcrop\n        signal = signal[:, :, cut_inf:cut_sup] #11250 * 32 * 282\n        #dt = axis_info['AIRS-CH0-integration_time'].dropna().values\n        dt = np.ones(len(signal))*0.1\n        dt[1::2] += 4.5 #@bilzard idea\n        linear_corr = linear_corr[:, :, cut_inf:cut_sup]\n        dark_frame = dark_frame[:, cut_inf:cut_sup]\n        dead_frame = dead_frame[:, cut_inf:cut_sup]\n        flat_frame = flat_frame[:, cut_inf:cut_sup]\n        hot = hot[:, cut_inf:cut_sup]\n    else:\n        dt = np.ones(len(signal))*0.1\n        dt[1::2] += 0.1\n        \n    signal = signal.clip(0) #@graySnow idea\n    linear_corr_signal = apply_linear_corr(linear_corr, signal)\n    signal = clean_dark(linear_corr_signal, dark_frame, dt)\n    \n    flat = flat_frame.reshape(sensor_sizes_dict[sensor][1])\n    flat[dead_frame.reshape(sensor_sizes_dict[sensor][1])] = np.nan\n    flat[hot.reshape(sensor_sizes_dict[sensor][1])] = np.nan\n    signal = signal / flat\n    \n    # 空間方向のcrop\n    if sensor == \"FGS1\":\n        signal = signal[:,10:22,10:22]\n        signal = signal.reshape(sensor_sizes_dict[sensor][0][0],144)\n    else:\n        signal = signal[:,10:22,:]\n\n    mean_signal = np.nanmean(signal, axis=1) # mean over the 32*32(FGS1) or 32(CH0) pixels\n    cds_signal = (mean_signal[1::2] - mean_signal[0::2])\n    \n    # 時間方向のbinning\n    binned = np.zeros((binned_dict[sensor]))\n    for j in range(cds_signal.shape[0] // binning):\n        binned[j] = cds_signal[j*binning:j*binning+binning].mean(axis=0)\n               \n    if sensor == \"FGS1\":\n        binned = binned.reshape((binned.shape[0],1))\n\n    if sensor != \"FGS1\":\n        # 波長方向をsliding windowによる平均化\n        # 周りのセンサーでnanのものがあったとしてもnanにはしていない\n        # 周りのセンサーでnanのものがある程度多かったらnanにするアイデアもある\n        if l > 0:\n            binned2 = np.zeros((11250 // binning // 2, 282))\n            for i in range(282):\n                binned2[:, i] = np.nanmean(binned[:, i:i+2*l+1], axis=1)\n                \n            binned = binned2\n\n        # reverse wavelength range\n        binned = binned[:, ::-1]\n\n    return binned\n\n\ndef preproc(dataset, adc_info, sensor, binning = 15):\n    planet_ids = adc_info.index\n\n    f = partial(read_planet, dataset=dataset, sensor=sensor, binning=binning)\n    z = list(zip(planet_ids, adc_info[f'{sensor}_adc_gain'], adc_info[f'{sensor}_adc_offset']))\n\n    if debug_num < 100:\n        z = z[:debug_num]\n\n    # feats = [f(zz) for zz in tqdm(z)]\n\n    with Pool() as p:\n        feats = list(tqdm(p.imap(f, z), total=len(z)))\n\n    return np.stack(feats)\n\n    \n# shapeは(num_planet, num_time, num_wavelength)\n# wavelengthの0番目はFGS1\ndata = np.concatenate([\n    preproc(mode, data_adc_info, \"FGS1\", 30*12),\n    preproc(mode, data_adc_info, \"AIRS-CH0\", 30)\n], axis=2)","metadata":{"execution":{"iopub.status.busy":"2024-10-19T06:19:22.763119Z","iopub.execute_input":"2024-10-19T06:19:22.76358Z","iopub.status.idle":"2024-10-19T06:20:42.004954Z","shell.execute_reply.started":"2024-10-19T06:19:22.76354Z","shell.execute_reply":"2024-10-19T06:20:42.003214Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### fit polynoms for each sample","metadata":{}},{"cell_type":"code","source":"def phase_detector(signal):\n    MIN = np.argmin(signal[30:140])+30\n    signal1 = signal[:MIN ]\n    signal2 = signal[MIN :]\n\n    first_derivative1 = np.gradient(signal1)\n    first_derivative1 /= first_derivative1.max()\n    first_derivative2 = np.gradient(signal2)\n    first_derivative2 /= first_derivative2.max()\n\n    phase1 = np.argmin(first_derivative1)\n    phase2 = np.argmax(first_derivative2) + MIN\n\n    return phase1, phase2\n\ndef polyfit(x, y, deg):\n    z = np.polyfit(x, y, deg)\n    p = np.poly1d(z)\n    y_pred = p(x)\n\n    return y_pred, p\n\ndelta = 2\n\ndef try_s(signal, p1, p2, deg, s):\n    x = list(range(signal.shape[0]-delta*4))\n    y = signal[:p1-delta].tolist() + (signal[p1+delta:p2 - delta] * (1 + s)).tolist() + signal[p2+delta:].tolist()\n    pred, p = polyfit(x, y, deg)\n    q = np.abs(pred - y).mean()\n\n    if s < 1e-4:\n        return q + 1e3\n\n    return q\n    \ndef calibrate_signal(data, return_all=False):\n    # dataのshapeは(num_time, num_wavelength)\n    p1, p2 = phase_detector(data.mean(axis=1))\n\n    s_l = []\n    x_l = []\n    y_l = []\n    pred_l = []\n    q_l = []\n    for i in range(283):\n        signal = data[:, i]\n        \n        if i == 0:\n            signal = data.mean(axis=1)\n\n        best_score = float('inf')\n        best_x, best_y, best_s, best_pred = None, None, None, None\n        for deg in range(4):\n            f = partial(try_s, signal, p1, p2, deg)\n            r = minimize(f, [0.0001], method = 'Nelder-Mead')\n            s = r.x[0]\n\n            x = list(range(signal.shape[0]-delta*4))\n            y = signal[:p1-delta].tolist() + (signal[p1+delta:p2 - delta] * (1 + s)).tolist() + signal[p2+delta:].tolist()\n            x = np.array(x)\n            y = np.array(y)\n\n            pred, p = polyfit(x, y, deg)\n            q = np.abs(pred - y).mean()\n            q = q/np.mean(signal[p1+delta:p2 - delta])*1000\n            pred = p(x)\n\n            if q < best_score:\n                best_score = q\n                best_x = x\n                best_y = y\n                best_s = s\n                best_pred = pred\n\n        s_l.append(best_s)\n        x_l.append(best_x)\n        y_l.append(best_y)\n        pred_l.append(best_pred)\n        q_l.append(best_score)\n\n    if return_all:\n        return np.array(s_l), np.array(q_l), x_l, y_l, pred_l\n    else:\n        return np.array(s_l), np.array(q_l)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-19T06:20:42.006953Z","iopub.execute_input":"2024-10-19T06:20:42.007373Z","iopub.status.idle":"2024-10-19T06:20:42.029018Z","shell.execute_reply.started":"2024-10-19T06:20:42.007323Z","shell.execute_reply":"2024-10-19T06:20:42.027443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"start = 34\nend = -39\n\nif debug_num < 100:\n    for n in range(debug_num):\n        s, q, x, y, y_new = calibrate_signal(data[n], return_all=True)\n        for i in range(283):\n            if i > 5 and i < 278:\n                continue\n\n            signal = data[n, :, i]\n            \n            if i == 0:\n                signal = data[n].mean(axis=1)\n\n            plt.scatter(np.arange(len(signal)), signal)\n            plt.scatter(x[i], y[i])\n            plt.scatter(x[i], y_new[i])\n            plt.show()\n            plt.hist(q, bins=30)\n            plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-19T06:20:42.032166Z","iopub.execute_input":"2024-10-19T06:20:42.03269Z","iopub.status.idle":"2024-10-19T06:24:22.494505Z","shell.execute_reply.started":"2024-10-19T06:20:42.032644Z","shell.execute_reply":"2024-10-19T06:24:22.493005Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with Pool() as p:\n    z = data\n\n    if debug_num < 100:\n        z = z[:debug_num]\n\n    all_s, all_q = zip(*tqdm(p.imap(calibrate_signal, z), total=len(z)))\n    all_s = np.stack(all_s)\n    all_q = np.stack(all_q)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-19T06:24:22.496235Z","iopub.execute_input":"2024-10-19T06:24:22.4967Z","iopub.status.idle":"2024-10-19T06:25:22.790358Z","shell.execute_reply.started":"2024-10-19T06:24:22.496658Z","shell.execute_reply":"2024-10-19T06:25:22.788744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_s = np.array(all_s)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-19T06:25:22.793111Z","iopub.execute_input":"2024-10-19T06:25:22.79366Z","iopub.status.idle":"2024-10-19T06:25:22.800455Z","shell.execute_reply.started":"2024-10-19T06:25:22.793579Z","shell.execute_reply":"2024-10-19T06:25:22.799112Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Sigma Estimation","metadata":{}},{"cell_type":"code","source":"if mode == \"train\" and not test_debug:\n    train_labels = pd.read_csv('/kaggle/input/neurips-ariel2024-train-fold/train_labels.csv')\n    wl_col = [f\"wl_{i+1}\" for i in range(283)]\n    train_labels.loc[:, wl_col] = (\n        train_labels.loc[:, wl_col]-pd.DataFrame(all_s, columns=wl_col)\n    ).abs()\n\n    if debug_num<100:\n        train_labels = train_labels.iloc[:debug_num]\n        data_adc_info = data_adc_info.iloc[:debug_num]\n    \n    oof = []\n    for fold in range(5):\n        preds = []\n        labels = []\n        for i in range(283):\n            train_mask = data_adc_info[\"fold\"].values!=fold\n            val_mask = data_adc_info[\"fold\"].values==fold\n            \n            X_train, X_val = all_q[train_mask, i], all_q[val_mask, i]\n            # print(train_mask.shape, train_labels.values.shape, train_labels.columns)\n            y_train, y_val = train_labels[train_mask].loc[:, f\"wl_{i+1}\"].values, train_labels.loc[val_mask].loc[:, f\"wl_{i+1}\"].values\n\n            model = LinearRegression()\n            model.fit(X_train.reshape(-1, 1), y_train)\n            y_pred = model.predict(X_val.reshape(-1, 1))\n\n            with open(f'model_{fold}_{i}.pkl', 'wb') as f:\n                pickle.dump(model, f)\n\n            labels.append(y_val.reshape(-1, 1))\n            preds.append(y_pred.reshape(-1, 1))\n\n        id = train_labels.loc[train_labels[\"fold\"]==fold, \"planet_id\"].values.reshape(-1, 1)\n        labels = np.concatenate(labels, axis=1)\n        preds = np.concatenate(preds, axis=1)\n        oof.append(np.concatenate([id, labels, preds], axis=1))\n\n    col = [f\"sigma_{i+1}\" for i in range(283)]\n    col_pred = [f\"sigma_{i+1}_pred\" for i in range(283)]\n\n    oof = np.concatenate(oof, axis=0)\n    oof = pd.DataFrame(oof, columns=[\"planet_id\"]+col+col_pred)\n    oof = oof.sort_values(\"planet_id\")\n    print(f\"MSE:{np.mean((oof.loc[:, col].values-oof.loc[:, col_pred].values)**2)}\")\n    if debug_num<100:\n        for i in range(debug_num):\n            d = np.abs(oof.loc[i, col]-oof.loc[i, col_pred])\n            plt.plot(np.arange(len(oof.loc[i, col])), oof.loc[i, col])\n            plt.plot(np.arange(len(oof.loc[i, col_pred])), oof.loc[i, col_pred])\n            plt.show()\n\n    data_sigma = oof.loc[:, col_pred].values\nelse:\n    preds = []\n    for i in range(283):\n        col_preds = []\n        for fold in range(5):\n            with open(f'model_{fold}_{i}.pkl', 'rb') as f:\n                model = pickle.load(f)\n\n            col_preds.append(model.predict(all_q[:, i].reshape(-1, 1)).reshape(-1, 1))\n        \n        col_preds = np.concatenate(col_preds, axis=1)\n        preds.append(np.mean(col_preds, axis=1).reshape(-1, 1))\n\n    preds = np.concatenate(preds, axis=1)\n    data_sigma = preds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-19T06:43:23.624777Z","iopub.execute_input":"2024-10-19T06:43:23.625259Z","iopub.status.idle":"2024-10-19T06:43:29.848491Z","shell.execute_reply.started":"2024-10-19T06:43:23.625222Z","shell.execute_reply":"2024-10-19T06:43:29.846997Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"I call the orange line \"starline\". This is probably what we would see if the planet weren't in the way.","metadata":{}},{"cell_type":"markdown","source":"### Making submission","metadata":{}},{"cell_type":"code","source":"ss = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/sample_submission.csv')\n\npreds = data_s.clip(0)\nsigmas = data_sigma.clip(0)\nsubmission = pd.DataFrame(np.concatenate([preds,sigmas], axis=1), columns=ss.columns[1:])\n\nif debug_num >= 100:\n    submission.index = data_adc_info.index\nelse:\n    submission.index = data_adc_info.index[:debug_num]\n\nif mode != \"train\":\n    !rm -rf ./*\n\nsubmission.to_csv('submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-19T06:43:40.497538Z","iopub.execute_input":"2024-10-19T06:43:40.498487Z","iopub.status.idle":"2024-10-19T06:43:40.547593Z","shell.execute_reply.started":"2024-10-19T06:43:40.498444Z","shell.execute_reply":"2024-10-19T06:43:40.546438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!head -n2 submission.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-19T06:42:36.39445Z","iopub.execute_input":"2024-10-19T06:42:36.394902Z","iopub.status.idle":"2024-10-19T06:42:37.466965Z","shell.execute_reply.started":"2024-10-19T06:42:36.394869Z","shell.execute_reply":"2024-10-19T06:42:37.465349Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ParticipantVisibleError(Exception):\n    pass\ndef ariel_score(\n        solution,\n        submission,\n        naive_mean,\n        naive_sigma,\n        sigma_true\n    ):\n    '''\n    This is a Gaussian Log Likelihood based metric. For a submission, which contains the predicted mean (x_hat) and variance (x_hat_std),\n    we calculate the Gaussian Log-likelihood (GLL) value to the provided ground truth (x). We treat each pair of x_hat,\n    x_hat_std as a 1D gaussian, meaning there will be 283 1D gaussian distributions, hence 283 values for each test spectrum,\n    the GLL value for one spectrum is the sum of all of them.\n\n    Inputs:\n        - solution: Ground Truth spectra (from test set)\n            - shape: (nsamples, n_wavelengths)\n        - submission: Predicted spectra and errors (from participants)\n            - shape: (nsamples, n_wavelengths*2)\n        naive_mean: (float) mean from the data set.\n        naive_sigma: (float) standard deviation from the data set.\n        sigma_true: (float) essentially sets the scale of the outputs.\n    '''\n\n    if submission.min() < 0:\n        raise ParticipantVisibleError('Negative values in the submission')\n\n    n_wavelengths = 283\n\n    y_pred = submission[:, :n_wavelengths]\n    # Set a non-zero minimum sigma pred to prevent division by zero errors.\n    sigma_pred = np.clip(submission[:, n_wavelengths:], a_min=10**-15, a_max=None)\n    y_true = solution\n\n    GLL_pred = np.sum(scipy.stats.norm.logpdf(y_true, loc=y_pred, scale=sigma_pred))\n    GLL_true = np.sum(scipy.stats.norm.logpdf(y_true, loc=y_true, scale=sigma_true * np.ones_like(y_true)))\n    GLL_mean = np.sum(scipy.stats.norm.logpdf(y_true, loc=naive_mean * np.ones_like(y_true), scale=naive_sigma * np.ones_like(y_true)))\n\n    #print(GLL_pred, GLL_true, GLL_mean)\n    submit_score = (GLL_pred - GLL_mean)/(GLL_true - GLL_mean)\n    return submit_score #float(np.clip(submit_score, 0.0, 1.0))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-19T06:42:41.978529Z","iopub.execute_input":"2024-10-19T06:42:41.979111Z","iopub.status.idle":"2024-10-19T06:42:41.991567Z","shell.execute_reply.started":"2024-10-19T06:42:41.979058Z","shell.execute_reply":"2024-10-19T06:42:41.990044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if mode == 'train':\n    data_labels = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/train_labels.csv', index_col='planet_id')\n    if debug_num >= 100:\n        print(ariel_score(\n                    data_labels.values,\n                    np.concatenate([preds, sigmas], axis=1), \n                    data_labels.values.mean(),\n                    data_labels.values.std(),\n                    sigma_true=1e-5))\n    else:\n        print(ariel_score(\n                    data_labels.values[:debug_num],\n                    np.concatenate([preds, sigmas], axis=1), \n                    data_labels.values.mean(),\n                    data_labels.values.std(),\n                    sigma_true=1e-5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-19T06:43:44.845217Z","iopub.execute_input":"2024-10-19T06:43:44.845682Z","iopub.status.idle":"2024-10-19T06:43:44.93215Z","shell.execute_reply.started":"2024-10-19T06:43:44.845646Z","shell.execute_reply":"2024-10-19T06:43:44.93083Z"}},"outputs":[],"execution_count":null}]}