{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":56537,"databundleVersionId":8877088,"sourceType":"competition"},{"sourceId":8849052,"sourceType":"datasetVersion","datasetId":5326404}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Postprocessing techniques\n\nThe main idea of this notebook is to make a compilation of all postprocessing techniques that I have been using, and which be applied to improve the score a little. None of this will get you to the first places, but given that a lot of them are already available in other notebooks, I decided to publish them all together here.","metadata":{}},{"cell_type":"code","source":"import os\nos.environ[\"KERAS_BACKEND\"] = \"jax\"\n\nimport gc\nimport numpy as np\nimport pandas as pd\n\nimport polars as pl\nimport matplotlib.pyplot as plt\n\nfrom sklearn import metrics\n\nfrom tqdm.notebook import tqdm\n\ndefault_color_1 = 'darkblue'\ndefault_color_2 = 'darkgreen'\ndefault_color_3 = 'darkred'\ndefault_color_4 = 'blue'\ndefault_color_5 = \"orange\"\ndefault_color_6 = \"yellow\"","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:13:58.417207Z","iopub.execute_input":"2024-07-03T13:13:58.417663Z","iopub.status.idle":"2024-07-03T13:13:58.429359Z","shell.execute_reply.started":"2024-07-03T13:13:58.417624Z","shell.execute_reply":"2024-07-03T13:13:58.428331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def is_interactive():\n    return 'runtime' in get_ipython().config.IPKernelApp.connection_file\n\nprint('Interactive?', is_interactive())","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:13:58.431291Z","iopub.execute_input":"2024-07-03T13:13:58.431685Z","iopub.status.idle":"2024-07-03T13:13:58.444612Z","shell.execute_reply.started":"2024-07-03T13:13:58.431654Z","shell.execute_reply":"2024-07-03T13:13:58.443402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA = \"/kaggle/input/leap-atmospheric-physics-ai-climsim\"\nDATA_TFREC = \"/kaggle/input/leap-train-tfrecords\"","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:13:58.4459Z","iopub.execute_input":"2024-07-03T13:13:58.446233Z","iopub.status.idle":"2024-07-03T13:13:58.45664Z","shell.execute_reply.started":"2024-07-03T13:13:58.446198Z","shell.execute_reply":"2024-07-03T13:13:58.455614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATS = pl.read_csv(os.path.join(DATA, \"test.csv\"), n_rows=1).select(pl.exclude('sample_id')).columns","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:13:58.459155Z","iopub.execute_input":"2024-07-03T13:13:58.459789Z","iopub.status.idle":"2024-07-03T13:13:58.49484Z","shell.execute_reply.started":"2024-07-03T13:13:58.45975Z","shell.execute_reply":"2024-07-03T13:13:58.493677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pl.read_csv(os.path.join(DATA, \"sample_submission.csv\"), n_rows=1)\nTARGETS = sample.select(pl.exclude('sample_id')).columns\nprint(len(TARGETS))","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:13:58.496237Z","iopub.execute_input":"2024-07-03T13:13:58.496622Z","iopub.status.idle":"2024-07-03T13:13:58.512875Z","shell.execute_reply.started":"2024-07-03T13:13:58.49659Z","shell.execute_reply":"2024-07-03T13:13:58.511564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Input data\n\nI am going to use the training data as an example, to show how each of these techniques improves the score when compared to the naive idea of replacing all target columns with the mean.","metadata":{}},{"cell_type":"code","source":"df_train = (\n    pl.scan_csv(\"/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv\")\n    .select(pl.exclude(\"sample_id\"))\n    .cast(pl.Float32)\n    .head(10000 if is_interactive() else 1000000)\n    .collect()\n).to_pandas()","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:13:58.51425Z","iopub.execute_input":"2024-07-03T13:13:58.514596Z","iopub.status.idle":"2024-07-03T13:13:59.021193Z","shell.execute_reply.started":"2024-07-03T13:13:58.514566Z","shell.execute_reply":"2024-07-03T13:13:59.019879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_x = df_train[FEATS]","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:13:59.023909Z","iopub.execute_input":"2024-07-03T13:13:59.024366Z","iopub.status.idle":"2024-07-03T13:13:59.041643Z","shell.execute_reply.started":"2024-07-03T13:13:59.024326Z","shell.execute_reply":"2024-07-03T13:13:59.040262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_y = df_train[TARGETS]","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:13:59.043538Z","iopub.execute_input":"2024-07-03T13:13:59.043909Z","iopub.status.idle":"2024-07-03T13:13:59.053214Z","shell.execute_reply.started":"2024-07-03T13:13:59.04388Z","shell.execute_reply":"2024-07-03T13:13:59.052111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_p = df_y.copy(deep=True)","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:13:59.054723Z","iopub.execute_input":"2024-07-03T13:13:59.055645Z","iopub.status.idle":"2024-07-03T13:13:59.07053Z","shell.execute_reply.started":"2024-07-03T13:13:59.055603Z","shell.execute_reply":"2024-07-03T13:13:59.069054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_y = df_y.mean()","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:13:59.074213Z","iopub.execute_input":"2024-07-03T13:13:59.074602Z","iopub.status.idle":"2024-07-03T13:13:59.086086Z","shell.execute_reply.started":"2024-07-03T13:13:59.074569Z","shell.execute_reply":"2024-07-03T13:13:59.084801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here, we fill every target column with the mean.","metadata":{}},{"cell_type":"code","source":"for target in tqdm(TARGETS):\n    df_p[target] = mean_y[target]","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:13:59.087437Z","iopub.execute_input":"2024-07-03T13:13:59.087809Z","iopub.status.idle":"2024-07-03T13:13:59.158145Z","shell.execute_reply.started":"2024-07-03T13:13:59.087778Z","shell.execute_reply":"2024-07-03T13:13:59.157017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Declare a helper function to plot R2 per target","metadata":{}},{"cell_type":"code","source":"def plot_r2_scores(scores_valid, ptend_q0002_max=None):\n    plt.figure(figsize=(15,10))\n    plt.plot(scores_valid.clip(-1, 1))\n    limit = 60\n    plt.axvline(x=limit, color='red', linestyle='--')\n    plt.axvspan(limit, limit+12, ymin = -1, ymax = 1, color='y', alpha=0.5, lw=0)\n    limit = 120\n    plt.axvline(x=limit, color='red', linestyle='--')\n    plt.axvspan(limit, limit+12, ymin = -1, ymax = 1, color='y', alpha=0.5, lw=0)\n    limit = 180\n    plt.axvline(x=limit, color='red', linestyle='--')\n    plt.axvspan(limit, limit+12, ymin = -1, ymax = 1, color='y', alpha=0.5, lw=0)\n    limit = 240\n    plt.axvline(x=limit, color='red', linestyle='--')\n    plt.axvspan(limit, limit+12, ymin = -1, ymax = 1, color='y', alpha=0.5, lw=0)\n    limit = 300\n    plt.axvline(x=limit, color='red', linestyle='--')\n    plt.axvspan(limit, limit+12, ymin = -1, ymax = 1, color='y', alpha=0.5, lw=0)\n    limit = 360\n    plt.axvline(x=limit, color='red', linestyle='--')\n    if ptend_q0002_max is not None:\n        plt.axvline(x=120+ptend_q0002_max, color='orange', linestyle='--')","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:13:59.159382Z","iopub.execute_input":"2024-07-03T13:13:59.159707Z","iopub.status.idle":"2024-07-03T13:13:59.170235Z","shell.execute_reply.started":"2024-07-03T13:13:59.159681Z","shell.execute_reply":"2024-07-03T13:13:59.169105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Initial score\n\nThis is the score if we just replace with the mean.","metadata":{}},{"cell_type":"code","source":"scores_valid = np.array([metrics.r2_score(df_y.values[:, i], df_p.values[:, i]) for i in range(len(TARGETS))])\nprint(f\"Validation score: {np.mean(scores_valid.clip(0, 1))}\")\nplot_r2_scores(scores_valid.clip(-1, 1))","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:13:59.171513Z","iopub.execute_input":"2024-07-03T13:13:59.171825Z","iopub.status.idle":"2024-07-03T13:14:01.322572Z","shell.execute_reply.started":"2024-07-03T13:13:59.171798Z","shell.execute_reply":"2024-07-03T13:14:01.321147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Zero out columns according to the weights","metadata":{}},{"cell_type":"code","source":"for target in tqdm(TARGETS):\n    if sample[target].to_numpy()[0] == 0.:\n        df_y.loc[:,target] = 0.\n        df_p.loc[:,target] = 0.\nscores_valid = np.array([metrics.r2_score(df_y.values[:, i], df_p.values[:, i]) for i in range(len(TARGETS))])\nprint(f\"Validation score: {np.mean(scores_valid.clip(0, 1))}\")\nplot_r2_scores(scores_valid.clip(-1, 1))","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:14:01.323821Z","iopub.execute_input":"2024-07-03T13:14:01.324163Z","iopub.status.idle":"2024-07-03T13:14:03.595244Z","shell.execute_reply.started":"2024-07-03T13:14:01.324127Z","shell.execute_reply":"2024-07-03T13:14:03.593825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Clipping ptend_q000* targets based on their respective input variable","metadata":{}},{"cell_type":"code","source":"for group in [\"q0001\",\"q0002\",\"q0003\"]:\n    for idx in range(60):\n        df_p.loc[:,f\"ptend_{group}_{idx}\"] = np.maximum(df_p[f\"ptend_{group}_{idx}\"],-df_x[f\"state_{group}_{idx}\"].to_numpy() / 1200.)\nscores_valid = np.array([metrics.r2_score(df_y.values[:, i], df_p.values[:, i]) for i in range(len(TARGETS))])\nprint(f\"Validation score: {np.mean(scores_valid.clip(0, 1))}\")\nplot_r2_scores(scores_valid.clip(-1, 1))","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:14:03.596662Z","iopub.execute_input":"2024-07-03T13:14:03.597012Z","iopub.status.idle":"2024-07-03T13:14:05.80049Z","shell.execute_reply.started":"2024-07-03T13:14:03.596983Z","shell.execute_reply":"2024-07-03T13:14:05.799195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ptend_q0002_* replacement","metadata":{}},{"cell_type":"code","source":"ptend_q0002_min = 12\nptend_q0002_max = 28\nfor idx in range(ptend_q0002_min, ptend_q0002_max):\n    df_p.loc[:,f\"ptend_q0002_{idx}\"] = -df_x[f\"state_q0002_{idx}\"].to_numpy() / 1200.\nscores_valid = np.array([metrics.r2_score(df_y.values[:, i], df_p.values[:, i]) for i in range(len(TARGETS))])\nprint(f\"Validation score: {np.mean(scores_valid.clip(0, 1))}\")\nplot_r2_scores(scores_valid.clip(-1, 1), ptend_q0002_max)","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:14:05.802025Z","iopub.execute_input":"2024-07-03T13:14:05.802518Z","iopub.status.idle":"2024-07-03T13:14:07.973117Z","shell.execute_reply.started":"2024-07-03T13:14:05.802474Z","shell.execute_reply":"2024-07-03T13:14:07.97173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Liquid-ice Cloud Partition\n\nTaken from [this paper](https://arxiv.org/pdf/2407.00124), see section 2.3.1","metadata":{}},{"cell_type":"code","source":"ICE_THRESHOLD = 253.16\nLIQUID_THRESHOLD = 273.16\nfor i in tqdm(range(60)):\n    new_t = df_x[f\"state_t_{i}\"] + df_y[f\"ptend_t_{i}\"]*1200.\n    \n    # I want to set the next df_test[f\"state_q0002_{i}\"] to zero!\n    if sample[f\"ptend_q0002_{i}\"].to_numpy()[0] != 0. and i >= ptend_q0002_max:\n        df_p.loc[new_t < ICE_THRESHOLD, f\"ptend_q0002_{i}\"] = -df_x[f\"state_q0002_{i}\"][new_t < ICE_THRESHOLD]/1200.\n    \n    # I want to set the next df_test[f\"state_q0003_{i}\"] to zero!\n    if sample[f\"ptend_q0003_{i}\"].to_numpy()[0] != 0.:\n        df_p.loc[new_t > LIQUID_THRESHOLD, f\"ptend_q0003_{i}\"] = -df_x[f\"state_q0003_{i}\"][new_t > LIQUID_THRESHOLD]/1200.","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:14:07.974647Z","iopub.execute_input":"2024-07-03T13:14:07.975013Z","iopub.status.idle":"2024-07-03T13:14:08.200638Z","shell.execute_reply.started":"2024-07-03T13:14:07.97498Z","shell.execute_reply":"2024-07-03T13:14:08.199279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_valid = np.array([metrics.r2_score(df_y.values[:, i], df_p.values[:, i]) for i in range(len(TARGETS))])\nprint(f\"Validation score: {np.mean(scores_valid.clip(0, 1))}\")\nplot_r2_scores(scores_valid.clip(-1, 1), ptend_q0002_max)","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:14:08.202311Z","iopub.execute_input":"2024-07-03T13:14:08.202682Z","iopub.status.idle":"2024-07-03T13:14:10.432194Z","shell.execute_reply.started":"2024-07-03T13:14:08.202649Z","shell.execute_reply":"2024-07-03T13:14:10.430623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ptend_q0003_* replacement\n\nI have seen some people using this, but it only works when working with small subsets of the dataset. Submitting with this replacement hurts my LB score.","metadata":{}},{"cell_type":"code","source":"ptend_q0003_min = 12\nptend_q0003_max = 16\nfor idx in range(ptend_q0003_min, ptend_q0003_max):\n    df_p.loc[:,f\"ptend_q0003_{idx}\"] = -df_x[f\"state_q0003_{idx}\"].to_numpy() / 1200.","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:14:11.873075Z","iopub.execute_input":"2024-07-03T13:14:11.873578Z","iopub.status.idle":"2024-07-03T13:14:11.880922Z","shell.execute_reply.started":"2024-07-03T13:14:11.873535Z","shell.execute_reply":"2024-07-03T13:14:11.879821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_valid = np.array([metrics.r2_score(df_y.values[:, i], df_p.values[:, i]) for i in range(len(TARGETS))])\nprint(f\"Validation score: {np.mean(scores_valid.clip(0, 1))}\")\nplot_r2_scores(scores_valid.clip(-1, 1), ptend_q0002_max)","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:14:11.882214Z","iopub.execute_input":"2024-07-03T13:14:11.882601Z","iopub.status.idle":"2024-07-03T13:14:14.031764Z","shell.execute_reply.started":"2024-07-03T13:14:11.88257Z","shell.execute_reply":"2024-07-03T13:14:14.030522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Removing clouds above stable tropopause level\n\nTaken from [this paper](https://arxiv.org/pdf/2407.00124), see section 2.3.2. STILL PENDING CORRECT IMPLEMENTATION.","metadata":{}},{"cell_type":"code","source":"#!pip install --upgrade netCDF4","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:14:14.033399Z","iopub.execute_input":"2024-07-03T13:14:14.033846Z","iopub.status.idle":"2024-07-03T13:14:14.039818Z","shell.execute_reply.started":"2024-07-03T13:14:14.033806Z","shell.execute_reply":"2024-07-03T13:14:14.038478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nimport xarray as xr\n\ndef get_pressure_thickness(a, b, p0, ps):\n    p = (a * p0).reshape(-1, 1) + (b.reshape(-1, 1) * ps)\n    return np.diff(p, axis=0).T\n\ndef get_pressure_at_interface(a, b, p0, ps):\n    p = (a * p0).reshape(-1, 1) + (b.reshape(-1, 1) * ps)\n    return p.T[:,:60]\n\ndef load_initial_params():\n    # ref: https://github.com/leap-stc/ClimSim/blob/main/grid_info/ClimSim_low-res_grid-info.nc\n    initial_conditions = \"/kaggle/input/climsim-low-res-grid/ClimSim_low-res_grid-info.nc\"\n    with xr.open_dataset(initial_conditions, engine=\"netcdf4\") as inic:\n        a = inic[\"hyai\"].to_numpy()\n        b = inic[\"hybi\"].to_numpy()\n        p0 = inic[\"P0\"].to_numpy()\n    return a, b, p0\n\na, b, p0 = load_initial_params()\ninterfaces_p = get_pressure_at_interface(a, b, p0, df_x[\"state_ps\"].values)\n'''","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:14:14.0413Z","iopub.execute_input":"2024-07-03T13:14:14.041758Z","iopub.status.idle":"2024-07-03T13:14:14.054283Z","shell.execute_reply.started":"2024-07-03T13:14:14.041713Z","shell.execute_reply":"2024-07-03T13:14:14.0531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nPRESSURE_THRESHOLD = 400*100\nGRADIENT_THRESHOLD = 1\n\nfor i in tqdm(range(len(df_y))):\n    new_t = []\n    for j in range(60):\n        new_t.append(df_x.loc[i,f\"state_t_{j}\"] + df_y.loc[i,f\"ptend_t_{j}\"]*1200.)\n    new_t = np.asarray(new_t)\n    t_gradient = np.diff(new_t)\n    \n    tropopause_level = np.intersect1d(np.argwhere(interfaces_p[i] < PRESSURE_THRESHOLD),np.argwhere(t_gradient > GRADIENT_THRESHOLD)).max(initial=0)\n    for j in range(tropopause_level):\n        feat_col = f\"state_q0002_{j}\"\n        target_col = f\"ptend_q0002_{j}\"\n        if sample[target_col].to_numpy()[0] != 0. and j >= ptend_q0002_max:\n            df_p.loc[i,target_col] = -df_x.loc[i,feat_col]/1200.\n            \n        feat_col = f\"state_q0003_{j}\"\n        target_col = f\"ptend_q0003_{j}\"\n        if sample[target_col].to_numpy()[0] != 0.:\n            df_p.loc[i,target_col] = -df_x.loc[i,feat_col]/1200.\n'''","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:14:14.055784Z","iopub.execute_input":"2024-07-03T13:14:14.056215Z","iopub.status.idle":"2024-07-03T13:14:14.07111Z","shell.execute_reply.started":"2024-07-03T13:14:14.056177Z","shell.execute_reply":"2024-07-03T13:14:14.069877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nscores_valid = np.array([metrics.r2_score(df_y.values[:, i], df_p.values[:, i]) for i in range(len(TARGETS))])\nprint(f\"Validation score: {np.mean(scores_valid.clip(0, 1))}\")\nplot_r2_scores(scores_valid.clip(-1, 1), ptend_q0002_max)\n'''","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:14:14.072701Z","iopub.execute_input":"2024-07-03T13:14:14.073131Z","iopub.status.idle":"2024-07-03T13:14:14.086084Z","shell.execute_reply.started":"2024-07-03T13:14:14.073091Z","shell.execute_reply":"2024-07-03T13:14:14.084841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### No clouds? No tendency to generate clouds!\n\nThis part here is also my proposal to implement something mentioned in [this paper](https://arxiv.org/pdf/2407.00124) in Appendix F. If a column has a clear sky (no cloud), the ptend_q0001 and ptend_q0002 could *maybe* be set to zero? I dont know, it doesn't work yet.","metadata":{}},{"cell_type":"code","source":"#MIXING_RATIO_THRESHOLD = 1e-15\n#for i in tqdm(range(60)):\n#    \n#    prev_cloud_mixing_ratio = df_x[f\"state_q0002_{i}\"] + df_x[f\"state_q0003_{i}\"]\n#    if sample[f\"ptend_q0002_{i}\"].to_numpy()[0] != 0. and i > ptend_q0002_max:\n#        df_p.loc[prev_cloud_mixing_ratio < MIXING_RATIO_THRESHOLD, f\"ptend_q0002_{i}\"] = 0.\n#    \n#    if sample[f\"ptend_q0003_{i}\"].to_numpy()[0] != 0.:\n#        df_p.loc[prev_cloud_mixing_ratio < MIXING_RATIO_THRESHOLD, f\"ptend_q0003_{i}\"] = 0.","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:14:14.126939Z","iopub.execute_input":"2024-07-03T13:14:14.127361Z","iopub.status.idle":"2024-07-03T13:14:14.138989Z","shell.execute_reply.started":"2024-07-03T13:14:14.127324Z","shell.execute_reply":"2024-07-03T13:14:14.13781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#scores_valid = np.array([metrics.r2_score(df_y.values[:, i], df_p.values[:, i]) for i in range(len(TARGETS))])\n#print(f\"Validation score: {np.mean(scores_valid.clip(0, 1))}\")\n#plot_r2_scores(scores_valid.clip(-1, 1), ptend_q0002_max)","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:14:14.140296Z","iopub.execute_input":"2024-07-03T13:14:14.140703Z","iopub.status.idle":"2024-07-03T13:14:14.152019Z","shell.execute_reply.started":"2024-07-03T13:14:14.140671Z","shell.execute_reply":"2024-07-03T13:14:14.15054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#for i in tqdm(range(15)):\n#    df_p.loc[:,f\"ptend_u_{i}\"] = 0.\n#    df_p.loc[:,f\"ptend_v_{i}\"] = 0.\n#scores_valid = np.array([metrics.r2_score(df_y.values[:, i], df_p.values[:, i]) for i in range(len(TARGETS))])\n#print(f\"Validation score: {np.mean(scores_valid.clip(0, 1))}\")\n#plot_r2_scores(scores_valid.clip(-1, 1))","metadata":{"execution":{"iopub.status.busy":"2024-07-03T13:14:14.153407Z","iopub.execute_input":"2024-07-03T13:14:14.15383Z","iopub.status.idle":"2024-07-03T13:14:14.164253Z","shell.execute_reply.started":"2024-07-03T13:14:14.153797Z","shell.execute_reply":"2024-07-03T13:14:14.163202Z"},"trusted":true},"execution_count":null,"outputs":[]}]}