{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook, we make a submission using sample means of targets. The goal is:\n\n- Understanding the submission format.\n- Understanding the competition metric.","metadata":{}},{"cell_type":"markdown","source":"# Setup","metadata":{}},{"cell_type":"code","source":"import os","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:32.949489Z","iopub.execute_input":"2024-04-23T03:03:32.950906Z","iopub.status.idle":"2024-04-23T03:03:33.00212Z","shell.execute_reply.started":"2024-04-23T03:03:32.950848Z","shell.execute_reply":"2024-04-23T03:03:33.000257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IS_INTERACTIVE = (os.environ.get('KAGGLE_KERNEL_RUN_TYPE', '') == 'Interactive')","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:33.14104Z","iopub.execute_input":"2024-04-23T03:03:33.141435Z","iopub.status.idle":"2024-04-23T03:03:33.148515Z","shell.execute_reply.started":"2024-04-23T03:03:33.141398Z","shell.execute_reply":"2024-04-23T03:03:33.146624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load dataset","metadata":{}},{"cell_type":"code","source":"import polars as pl\nimport polars.selectors as cs","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:34.188914Z","iopub.execute_input":"2024-04-23T03:03:34.189416Z","iopub.status.idle":"2024-04-23T03:03:34.511513Z","shell.execute_reply.started":"2024-04-23T03:03:34.189379Z","shell.execute_reply":"2024-04-23T03:03:34.510074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We use only a few samples.","metadata":{}},{"cell_type":"code","source":"NUM_TRAIN_SAMPLES = 10_000 if IS_INTERACTIVE else 1_000_000","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:34.904701Z","iopub.execute_input":"2024-04-23T03:03:34.905316Z","iopub.status.idle":"2024-04-23T03:03:34.913802Z","shell.execute_reply.started":"2024-04-23T03:03:34.905266Z","shell.execute_reply":"2024-04-23T03:03:34.912394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lf_train = pl.scan_csv(\n    '/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv'\n).head(NUM_TRAIN_SAMPLES).drop('sample_id')","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:35.277155Z","iopub.execute_input":"2024-04-23T03:03:35.277644Z","iopub.status.idle":"2024-04-23T03:03:35.374915Z","shell.execute_reply.started":"2024-04-23T03:03:35.277609Z","shell.execute_reply":"2024-04-23T03:03:35.37315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(lf_train.head().collect())","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:35.482344Z","iopub.execute_input":"2024-04-23T03:03:35.48295Z","iopub.status.idle":"2024-04-23T03:03:35.857028Z","shell.execute_reply.started":"2024-04-23T03:03:35.482902Z","shell.execute_reply":"2024-04-23T03:03:35.855251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGET_SELECTORS = [\n    cs.starts_with('ptend_'),\n    cs.starts_with('cam_out_'),\n]","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:35.940803Z","iopub.execute_input":"2024-04-23T03:03:35.941471Z","iopub.status.idle":"2024-04-23T03:03:35.950926Z","shell.execute_reply.started":"2024-04-23T03:03:35.941391Z","shell.execute_reply":"2024-04-23T03:03:35.948844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = lf_train.drop(*TARGET_SELECTORS).collect().to_pandas()\nY_train = lf_train.select(*TARGET_SELECTORS).collect().to_pandas()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:36.229507Z","iopub.execute_input":"2024-04-23T03:03:36.229923Z","iopub.status.idle":"2024-04-23T03:03:38.251159Z","shell.execute_reply.started":"2024-04-23T03:03:36.22989Z","shell.execute_reply":"2024-04-23T03:03:38.249605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(X_train)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:38.256343Z","iopub.execute_input":"2024-04-23T03:03:38.257033Z","iopub.status.idle":"2024-04-23T03:03:38.319229Z","shell.execute_reply.started":"2024-04-23T03:03:38.256993Z","shell.execute_reply":"2024-04-23T03:03:38.317514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(Y_train)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:38.321012Z","iopub.execute_input":"2024-04-23T03:03:38.321468Z","iopub.status.idle":"2024-04-23T03:03:38.363834Z","shell.execute_reply.started":"2024-04-23T03:03:38.321405Z","shell.execute_reply":"2024-04-23T03:03:38.362269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`sample_submission.csv` contains the weight of every target which will be used for making submission and CV.","metadata":{}},{"cell_type":"code","source":"lf_sample_submission = pl.scan_csv('/kaggle/input/leap-atmospheric-physics-ai-climsim/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:38.366528Z","iopub.execute_input":"2024-04-23T03:03:38.366955Z","iopub.status.idle":"2024-04-23T03:03:38.397471Z","shell.execute_reply.started":"2024-04-23T03:03:38.366921Z","shell.execute_reply":"2024-04-23T03:03:38.39588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(lf_sample_submission.head().collect())","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:38.401088Z","iopub.execute_input":"2024-04-23T03:03:38.401938Z","iopub.status.idle":"2024-04-23T03:03:38.542669Z","shell.execute_reply.started":"2024-04-23T03:03:38.401883Z","shell.execute_reply":"2024-04-23T03:03:38.540985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGET_SCHEMA = lf_sample_submission.drop('sample_id').schema","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:39.093665Z","iopub.execute_input":"2024-04-23T03:03:39.094158Z","iopub.status.idle":"2024-04-23T03:03:39.105788Z","shell.execute_reply.started":"2024-04-23T03:03:39.094122Z","shell.execute_reply":"2024-04-23T03:03:39.104089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGET_WEIGHTS = lf_sample_submission.drop('sample_id').head(1).collect().to_numpy()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:39.343269Z","iopub.execute_input":"2024-04-23T03:03:39.344333Z","iopub.status.idle":"2024-04-23T03:03:39.374176Z","shell.execute_reply.started":"2024-04-23T03:03:39.34429Z","shell.execute_reply":"2024-04-23T03:03:39.372524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(TARGET_WEIGHTS)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:39.517437Z","iopub.execute_input":"2024-04-23T03:03:39.518002Z","iopub.status.idle":"2024-04-23T03:03:39.534375Z","shell.execute_reply.started":"2024-04-23T03:03:39.517966Z","shell.execute_reply":"2024-04-23T03:03:39.532705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Metric","metadata":{}},{"cell_type":"markdown","source":"The competition metric is R2 on weighted targets. If I understand correctly, it is\n\n$$R^{2}(y \\odot w, \\hat{y} \\odot w)$$\n\nwhere $R^{2}$ is the R2 score, $w$ is the target weights, and $\\odot$ is the element-wise multiplication.","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import r2_score, make_scorer\nfrom sklearn.model_selection import cross_validate, KFold\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:40.909505Z","iopub.execute_input":"2024-04-23T03:03:40.909981Z","iopub.status.idle":"2024-04-23T03:03:41.719653Z","shell.execute_reply.started":"2024-04-23T03:03:40.909946Z","shell.execute_reply":"2024-04-23T03:03:41.718277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_metric(y_true, y_pred):\n    target_weights = np.asarray(TARGET_WEIGHTS)\n    y_true = np.asarray(y_true)\n    y_pred = np.asarray(y_pred)\n    y_true = y_true * target_weights\n    y_pred = y_pred * target_weights\n    return r2_score(y_true, y_pred)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:41.722166Z","iopub.execute_input":"2024-04-23T03:03:41.722648Z","iopub.status.idle":"2024-04-23T03:03:41.734178Z","shell.execute_reply.started":"2024-04-23T03:03:41.722609Z","shell.execute_reply":"2024-04-23T03:03:41.732637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"custom_scorer = make_scorer(custom_metric)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:41.736275Z","iopub.execute_input":"2024-04-23T03:03:41.736732Z","iopub.status.idle":"2024-04-23T03:03:41.744973Z","shell.execute_reply.started":"2024-04-23T03:03:41.736692Z","shell.execute_reply":"2024-04-23T03:03:41.743524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_cv(estimator, X = X_train, Y = Y_train,\n              cv = KFold(shuffle = True, random_state = 0)):\n    result = cross_validate(\n        estimator, X, Y,\n        cv = cv,\n        scoring = {\n            'custom_metric': custom_scorer,\n        },\n        error_score = 'raise',\n    )\n    return pl.from_dict(result).with_row_index('fold')","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:41.790026Z","iopub.execute_input":"2024-04-23T03:03:41.790545Z","iopub.status.idle":"2024-04-23T03:03:41.799465Z","shell.execute_reply.started":"2024-04-23T03:03:41.790506Z","shell.execute_reply":"2024-04-23T03:03:41.797803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dummy regressor","metadata":{}},{"cell_type":"code","source":"from sklearn.dummy import DummyRegressor","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:42.213267Z","iopub.execute_input":"2024-04-23T03:03:42.213753Z","iopub.status.idle":"2024-04-23T03:03:42.223256Z","shell.execute_reply.started":"2024-04-23T03:03:42.213719Z","shell.execute_reply":"2024-04-23T03:03:42.22194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dummy_0 = DummyRegressor()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:42.676791Z","iopub.execute_input":"2024-04-23T03:03:42.677203Z","iopub.status.idle":"2024-04-23T03:03:42.683697Z","shell.execute_reply.started":"2024-04-23T03:03:42.677173Z","shell.execute_reply":"2024-04-23T03:03:42.68208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv_dummy_0 = custom_cv(dummy_0)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:43.194148Z","iopub.execute_input":"2024-04-23T03:03:43.194658Z","iopub.status.idle":"2024-04-23T03:03:43.74975Z","shell.execute_reply.started":"2024-04-23T03:03:43.194621Z","shell.execute_reply":"2024-04-23T03:03:43.748554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(cv_dummy_0)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:43.936297Z","iopub.execute_input":"2024-04-23T03:03:43.936823Z","iopub.status.idle":"2024-04-23T03:03:43.947533Z","shell.execute_reply.started":"2024-04-23T03:03:43.936784Z","shell.execute_reply":"2024-04-23T03:03:43.945877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(cv_dummy_0.select(pl.col('test_custom_metric').mean()))","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:45.388212Z","iopub.execute_input":"2024-04-23T03:03:45.388674Z","iopub.status.idle":"2024-04-23T03:03:45.408055Z","shell.execute_reply.started":"2024-04-23T03:03:45.388639Z","shell.execute_reply":"2024-04-23T03:03:45.406463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will get `0.16690` on LB.","metadata":{}},{"cell_type":"markdown","source":"Note that the expected R2 score of mean prediction is `0` but we get higher score. This is because some targets have zero weight and the R2 score on a zero weight target is always `1` (in sklearn implementation). Indeed, the CV and LB scores are close to the proportion of zero weight targets.","metadata":{}},{"cell_type":"code","source":"display(lf_sample_submission.head(1).drop('sample_id').select(\n    pl.concat_list(pl.all() == 0).list.mean().alias('zero_weight_proportion'),\n).collect())","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:06:45.562089Z","iopub.execute_input":"2024-04-23T03:06:45.562706Z","iopub.status.idle":"2024-04-23T03:06:45.599094Z","shell.execute_reply.started":"2024-04-23T03:06:45.562663Z","shell.execute_reply":"2024-04-23T03:06:45.597704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"final_estimator = dummy_0.fit(X_train, Y_train)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T01:02:28.542144Z","iopub.execute_input":"2024-04-23T01:02:28.543201Z","iopub.status.idle":"2024-04-23T01:02:28.557489Z","shell.execute_reply.started":"2024-04-23T01:02:28.54314Z","shell.execute_reply":"2024-04-23T01:02:28.55628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lf_test = pl.scan_csv('/kaggle/input/leap-atmospheric-physics-ai-climsim/test.csv')\nif IS_INTERACTIVE:\n    lf_test = lf_test.head(10_000)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T01:02:37.863404Z","iopub.execute_input":"2024-04-23T01:02:37.863762Z","iopub.status.idle":"2024-04-23T01:02:37.89758Z","shell.execute_reply.started":"2024-04-23T01:02:37.863736Z","shell.execute_reply":"2024-04-23T01:02:37.896224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(lf_test.head().collect())","metadata":{"execution":{"iopub.status.busy":"2024-04-23T01:02:39.103377Z","iopub.execute_input":"2024-04-23T01:02:39.103768Z","iopub.status.idle":"2024-04-23T01:02:39.246422Z","shell.execute_reply.started":"2024-04-23T01:02:39.10374Z","shell.execute_reply":"2024-04-23T01:02:39.245465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_predict_1(estimator, df):\n    Y = estimator.predict(df.to_pandas())\n    target_weights = np.asarray(TARGET_WEIGHTS)\n    Y = Y * target_weights\n    return pl.from_numpy(Y, schema = TARGET_SCHEMA)\n\ndef custom_predict(estimator, lf):\n    Y = lf.drop('sample_id').map_batches(\n        (lambda df: custom_predict_1(estimator, df)),\n        schema = TARGET_SCHEMA,\n        streamable = True,\n    )\n    return pl.concat([lf.select('sample_id'), Y], how = 'horizontal')","metadata":{"execution":{"iopub.status.busy":"2024-04-23T01:02:45.638558Z","iopub.execute_input":"2024-04-23T01:02:45.639538Z","iopub.status.idle":"2024-04-23T01:02:45.646333Z","shell.execute_reply.started":"2024-04-23T01:02:45.639502Z","shell.execute_reply":"2024-04-23T01:02:45.644959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lf_pred = custom_predict(final_estimator, lf_test)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T01:02:49.479879Z","iopub.execute_input":"2024-04-23T01:02:49.480213Z","iopub.status.idle":"2024-04-23T01:02:49.490494Z","shell.execute_reply.started":"2024-04-23T01:02:49.48019Z","shell.execute_reply":"2024-04-23T01:02:49.48888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lf_pred.collect().write_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-23T01:02:50.782312Z","iopub.execute_input":"2024-04-23T01:02:50.782715Z","iopub.status.idle":"2024-04-23T01:02:51.750134Z","shell.execute_reply.started":"2024-04-23T01:02:50.782683Z","shell.execute_reply":"2024-04-23T01:02:51.748619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(pl.scan_csv('submission.csv').head().collect())","metadata":{"execution":{"iopub.status.busy":"2024-04-23T01:03:11.71419Z","iopub.execute_input":"2024-04-23T01:03:11.714575Z","iopub.status.idle":"2024-04-23T01:03:11.752086Z","shell.execute_reply.started":"2024-04-23T01:03:11.714553Z","shell.execute_reply":"2024-04-23T01:03:11.750919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}