{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.13"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":101849,"databundleVersionId":13093295,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":9629432,"sourceType":"datasetVersion","datasetId":5846888},{"sourceId":12844435,"sourceType":"datasetVersion","datasetId":8123780},{"sourceId":13109982,"sourceType":"datasetVersion","datasetId":8304548,"isSourceIdPinned":true},{"sourceId":13154167,"sourceType":"datasetVersion","datasetId":8334332},{"sourceId":562901,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":425925,"modelId":443413},{"sourceId":562906,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":425929,"modelId":443416}],"dockerImageVersionId":31090,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":25.428306,"end_time":"2025-09-19T07:12:03.82627","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-09-19T07:11:38.397964","version":"2.6.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"0221a12bf8b641a88c27a35a22782791":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"08dfcda2da794462b3ee882ccd02f395":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"0b8a058678604ebbbbbdc21cb6610a88":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_941a807a0d194965b7c51bceb9f95ac9","IPY_MODEL_236d4b4543f247c8aba6614c0bcfd867","IPY_MODEL_58555c7a27e247d482e21996c528c9fc"],"layout":"IPY_MODEL_0f49c451399942a08b0bc4ac57e9cf24","tabbable":null,"tooltip":null}},"0b9906a7b1ba4567babb37c2d60aa1f4":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"0cf46e951a77416881d5290737a4fa5a":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"0f49c451399942a08b0bc4ac57e9cf24":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"1421eb6bef4e431ca6e0260b61658934":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"14ae4578043d4771bef60ed3cbea931c":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_4c7bf3feae784d8eba4158ac99e586f0","placeholder":"​","style":"IPY_MODEL_2350cec3b6e94e5f9d134ceb728e3836","tabbable":null,"tooltip":null,"value":" 1/1 [00:01&lt;00:00,  1.95s/it]"}},"189fe9759ac44818b6f02937e2b286f2":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"1d1d7afb980b48d9994b27a732b5b369":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_90b006b9e7944f22a9d69dc984ede712","IPY_MODEL_e23a77085f114dafa9f74cbe01a763eb","IPY_MODEL_7296539cae04499aa1ab1d03eacb7dab"],"layout":"IPY_MODEL_d24efe355daa44d6b5178bcdd723fd0e","tabbable":null,"tooltip":null}},"212db83369c44a829d1d576ffdfdfa4b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_aefa454f0ae444898a4951ef5f85934f","IPY_MODEL_6c24188c13c44ffa8b393adaf5985f09","IPY_MODEL_590b6380947c4a9e91011350dad78521"],"layout":"IPY_MODEL_de6bcf4060574b48a821d3f4b7ddf226","tabbable":null,"tooltip":null}},"21cbb4523c5642edba4a5fe2ab711e77":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_e40ba11019ee4bdb87d1923a283bf0ba","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_08dfcda2da794462b3ee882ccd02f395","tabbable":null,"tooltip":null,"value":1}},"2350cec3b6e94e5f9d134ceb728e3836":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"236d4b4543f247c8aba6614c0bcfd867":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_c16112696bf94043b9c5cf33c94b83ad","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_454e4aabdbbf47ec9d430ef1ef114b88","tabbable":null,"tooltip":null,"value":1}},"24dc4562f6394bafb93c6ed818638916":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_44f2827e1cfc405eb82b62ef5dbc935f","placeholder":"​","style":"IPY_MODEL_8658c5da8cb04977aff947507178bb38","tabbable":null,"tooltip":null,"value":"QUEUEING TASKS | : 100%"}},"269462dbae06474e9ee44a3f9d228c61":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"2a3a314647af4bf78305d68c5bb54276":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_ba1c300814c244319a89907365905c40","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_9423a83484c643e2b610c564072711ac","tabbable":null,"tooltip":null,"value":1}},"2b82dc160b4948aab93a0a80105f3ed2":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"2d7f5dc948314103ab32dfe56546fd03":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_d4bd3f4ad9e34e1fb18cac2619ee6486","placeholder":"​","style":"IPY_MODEL_68b4ab438c2c420d8e7c1eaac7afbd68","tabbable":null,"tooltip":null,"value":"PROCESSING TASKS | : 100%"}},"30a1464f20a642f0b11b7091a9f37bb8":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"354005cbb8fb4c84a75112772a55306d":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_24dc4562f6394bafb93c6ed818638916","IPY_MODEL_21cbb4523c5642edba4a5fe2ab711e77","IPY_MODEL_90ef9b0099b04f7ebbc1e8fc221d1bd6"],"layout":"IPY_MODEL_849be3ebb49043d0a0c9cc287dc5a3d0","tabbable":null,"tooltip":null}},"44f2827e1cfc405eb82b62ef5dbc935f":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"454e4aabdbbf47ec9d430ef1ef114b88":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"4aecfd1e48a94a67a7666423c31193e7":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_2d7f5dc948314103ab32dfe56546fd03","IPY_MODEL_2a3a314647af4bf78305d68c5bb54276","IPY_MODEL_14ae4578043d4771bef60ed3cbea931c"],"layout":"IPY_MODEL_c89e08166c024dd3a0dd0427ba229082","tabbable":null,"tooltip":null}},"4c7bf3feae784d8eba4158ac99e586f0":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"512addfc7bde4e1d9060aeb8cf7121f5":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"58555c7a27e247d482e21996c528c9fc":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_0221a12bf8b641a88c27a35a22782791","placeholder":"​","style":"IPY_MODEL_30a1464f20a642f0b11b7091a9f37bb8","tabbable":null,"tooltip":null,"value":" 1/1 [00:03&lt;00:00,  3.29s/it]"}},"590b6380947c4a9e91011350dad78521":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_512addfc7bde4e1d9060aeb8cf7121f5","placeholder":"​","style":"IPY_MODEL_5b97b87685574e30aeb301c8daa25bf0","tabbable":null,"tooltip":null,"value":" 1/1 [00:00&lt;00:00, 116.42it/s]"}},"5a8d8fede7b147d99b7d992389f93d39":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"5ade02a1c1114e0dbc9e2aa6b5b9a94a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"5b97b87685574e30aeb301c8daa25bf0":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"68b4ab438c2c420d8e7c1eaac7afbd68":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"6c24188c13c44ffa8b393adaf5985f09":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_deeca099ce7d4b7abdf7d38cd4cc857f","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_5ade02a1c1114e0dbc9e2aa6b5b9a94a","tabbable":null,"tooltip":null,"value":1}},"6c4c8fe86e314382b00d72c34d49c3e3":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"7296539cae04499aa1ab1d03eacb7dab":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_b48ede0d236a4bfaa93d32e59fdfa3cc","placeholder":"​","style":"IPY_MODEL_b0e6f779c11d49d180ef7898ac8db463","tabbable":null,"tooltip":null,"value":" 1/1 [00:00&lt;00:00, 120.98it/s]"}},"7e95ba94c897445ab8e4f423dd21edf3":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"849be3ebb49043d0a0c9cc287dc5a3d0":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8658c5da8cb04977aff947507178bb38":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"8ca91ba236334f629b2e5bb5e3a00017":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8f6c667a843541cda0989163f93d692f":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"90b006b9e7944f22a9d69dc984ede712":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_2b82dc160b4948aab93a0a80105f3ed2","placeholder":"​","style":"IPY_MODEL_189fe9759ac44818b6f02937e2b286f2","tabbable":null,"tooltip":null,"value":"COLLECTING RESULTS | : 100%"}},"90ef9b0099b04f7ebbc1e8fc221d1bd6":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_1421eb6bef4e431ca6e0260b61658934","placeholder":"​","style":"IPY_MODEL_5a8d8fede7b147d99b7d992389f93d39","tabbable":null,"tooltip":null,"value":" 1/1 [00:00&lt;00:00, 106.84it/s]"}},"93157a829c464d17820735b94215f356":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_a8d91662839743cebcbe9093c539bb3b","placeholder":"​","style":"IPY_MODEL_cf92069fd2b84c71ba91ec2293079823","tabbable":null,"tooltip":null,"value":" 1/1 [00:00&lt;00:00, 124.80it/s]"}},"941a807a0d194965b7c51bceb9f95ac9":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_8f6c667a843541cda0989163f93d692f","placeholder":"​","style":"IPY_MODEL_ba2b21785a1b4e4ebd7743ecb6d5ef17","tabbable":null,"tooltip":null,"value":"PROCESSING TASKS | : 100%"}},"9423a83484c643e2b610c564072711ac":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"9d4b0186bd6e46a18ccd32d24d49e285":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_cd2488be75eb409a9efb2caedcedda7d","IPY_MODEL_af747d08a4db47e8969fd622d3ba9bc1","IPY_MODEL_93157a829c464d17820735b94215f356"],"layout":"IPY_MODEL_9e1848968a344197a2f802904c9b65c9","tabbable":null,"tooltip":null}},"9e1848968a344197a2f802904c9b65c9":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"a8d91662839743cebcbe9093c539bb3b":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"aefa454f0ae444898a4951ef5f85934f":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_8ca91ba236334f629b2e5bb5e3a00017","placeholder":"​","style":"IPY_MODEL_c52fbf3fd8cd4288b112c04ea561e608","tabbable":null,"tooltip":null,"value":"QUEUEING TASKS | : 100%"}},"af747d08a4db47e8969fd622d3ba9bc1":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_269462dbae06474e9ee44a3f9d228c61","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_0b9906a7b1ba4567babb37c2d60aa1f4","tabbable":null,"tooltip":null,"value":1}},"b0e6f779c11d49d180ef7898ac8db463":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"b48ede0d236a4bfaa93d32e59fdfa3cc":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"ba1c300814c244319a89907365905c40":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"ba2b21785a1b4e4ebd7743ecb6d5ef17":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"c16112696bf94043b9c5cf33c94b83ad":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"c52fbf3fd8cd4288b112c04ea561e608":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"c89e08166c024dd3a0dd0427ba229082":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"cd2488be75eb409a9efb2caedcedda7d":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_6c4c8fe86e314382b00d72c34d49c3e3","placeholder":"​","style":"IPY_MODEL_d548b62647564e2988b3bbfe4d496a71","tabbable":null,"tooltip":null,"value":"COLLECTING RESULTS | : 100%"}},"cf92069fd2b84c71ba91ec2293079823":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"d24efe355daa44d6b5178bcdd723fd0e":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"d4bd3f4ad9e34e1fb18cac2619ee6486":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"d548b62647564e2988b3bbfe4d496a71":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"de6bcf4060574b48a821d3f4b7ddf226":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"deeca099ce7d4b7abdf7d38cd4cc857f":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"e23a77085f114dafa9f74cbe01a763eb":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_0cf46e951a77416881d5290737a4fa5a","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_7e95ba94c897445ab8e4f423dd21edf3","tabbable":null,"tooltip":null,"value":1}},"e40ba11019ee4bdb87d1923a283bf0ba":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install --no-index --find-links=/kaggle/input/ariel-2024-pqdm pqdm > /dev/null","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2025-09-24T05:28:15.02394Z","iopub.execute_input":"2025-09-24T05:28:15.024233Z","iopub.status.idle":"2025-09-24T05:28:19.165757Z","shell.execute_reply.started":"2025-09-24T05:28:15.024211Z","shell.execute_reply":"2025-09-24T05:28:19.1648Z"},"papermill":{"duration":4.302424,"end_time":"2025-09-19T07:11:47.255904","exception":false,"start_time":"2025-09-19T07:11:42.95348","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn.functional as F\nimport multiprocessing as mp\nimport torch.nn as nn\nimport os\nimport matplotlib.pyplot as plt\nimport itertools\n\nfrom tqdm import tqdm\nfrom pqdm.threads import pqdm\nfrom astropy.stats import sigma_clip\nfrom scipy.optimize import minimize\nfrom torch.utils.data import DataLoader, TensorDataset, random_split\nfrom sklearn.preprocessing import StandardScaler\nfrom scipy.signal import savgol_filter\nfrom sklearn.metrics import mean_squared_error","metadata":{"execution":{"iopub.status.busy":"2025-09-24T05:28:19.167369Z","iopub.execute_input":"2025-09-24T05:28:19.167609Z","iopub.status.idle":"2025-09-24T05:28:25.238396Z","shell.execute_reply.started":"2025-09-24T05:28:19.167581Z","shell.execute_reply":"2025-09-24T05:28:25.237795Z"},"papermill":{"duration":8.611729,"end_time":"2025-09-19T07:11:55.870946","exception":false,"start_time":"2025-09-19T07:11:47.259217","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ROOT_PATH = \"/kaggle/input/ariel-data-challenge-2025\"\nMODE = \"test\"\n\nclass Config:\n    DATA_PATH = '/kaggle/input/ariel-data-challenge-2025'\n    DATASET = \"test\"\n\n    SCALE = 0.946\n    SIGMA = 0.00056\n    \n    CUT_INF = 39\n    CUT_SUP = 321\n    \n    SENSOR_CONFIG = {\n        \"AIRS-CH0\": {\n            \"raw_shape\": [11250, 32, 356],\n            \"calibrated_shape\": [1, 32, CUT_SUP - CUT_INF],\n            \"linear_corr_shape\": (6, 32, 356),\n            \"dt_pattern\": (0.1, 4.5), \n            \"binning\": 30\n        },\n        \"FGS1\": {\n            \"raw_shape\": [135000, 32, 32],\n            \"calibrated_shape\": [1, 32, 32],\n            \"linear_corr_shape\": (6, 32, 32),\n            \"dt_pattern\": (0.1, 0.1),\n            \"binning\": 30 * 12\n        }\n    }\n    \n    MODEL_PHASE_DETECTION_SLICE = slice(30, 140)\n    MODEL_OPTIMIZATION_DELTA = 11\n    MODEL_POLYNOMIAL_DEGREE = 3\n    \n    N_JOBS = 3\n\ndef _phase_detector_signal(signal, cfg):\n    sl = cfg.MODEL_PHASE_DETECTION_SLICE\n    min_idx = int(np.argmin(signal[sl])) + sl.start\n    s1 = signal[:min_idx]; s2 = signal[min_idx:]\n    \n    if s1.size < 3 or s2.size < 3:\n        return 0, len(signal) - 1\n    \n    g1 = np.gradient(s1); g1_max = np.max(g1) if np.size(g1) else 0.0\n    g2 = np.gradient(s2); g2_max = np.max(g2) if np.size(g2) else 0.0\n    \n    if g1_max != 0:\n        g1 /= g1_max\n    if g2_max != 0:\n        g2 /= g2_max\n    \n    phase1 = int(np.argmin(g1))\n    phase2 = int(np.argmax(g2)) + min_idx\n    \n    return phase1, phase2\n\ndef estimate_sigma_fgs(preprocessed_data, cfg):\n    sig_rel = []\n    delta = cfg.MODEL_OPTIMIZATION_DELTA\n    eps = 1e-12\n    \n    for single in preprocessed_data:\n        air_white = savgol_filter(single[:, 1:].mean(axis=1), 20, 2)\n        p1, p2 = _phase_detector_signal(air_white, cfg)\n        p1 = max(delta, p1)\n        p2 = min(len(air_white) - delta - 1, p2)\n\n        fgs = single[:, 0]\n        oot = (fgs[: p1 - delta] if p1 - delta > 0 else np.empty(0, fgs.dtype))\n        if p2 + delta < fgs.size:\n            oot = np.concatenate([oot, fgs[p2 + delta :]])\n        inn = fgs[p1 + delta : max(p1 + delta, p2 - delta)]\n\n        if oot.size == 0 or inn.size == 0:\n            sig_rel.append(np.nan); continue\n\n        n_oot, n_in = len(oot), len(inn)\n        var_oot = np.nanvar(oot, ddof=1)\n        var_in  = np.nanvar(inn, ddof=1)\n        oot_mean = float(np.nanmean(oot)) if np.isfinite(np.nanmean(oot)) else float(np.nanmean(fgs))\n        sigma_rel = np.sqrt(var_oot / max(n_oot,1) + var_in / max(n_in,1)) / max(oot_mean, eps)\n        sig_rel.append(sigma_rel)\n\n    s = np.asarray(sig_rel, dtype=float)\n    mask = np.isfinite(s) & (s > 0)\n    med = float(np.nanmedian(s[mask])) if mask.any() else 1.0\n\n    k = np.ones_like(s)\n    if med > 0 and np.isfinite(med):\n        k[mask] = np.sqrt(s[mask] / med)\n    \n    k = np.clip(k, 0.85, 1.30) \n    \n    return k * cfg.SIGMA * 1.04\n\ndef estimate_sigma_air(preprocessed_data, cfg):\n    \"\"\"Return sigma_air length N_planets.\"\"\"\n    sig_rel = []\n    delta = cfg.MODEL_OPTIMIZATION_DELTA\n    eps = 1e-12\n\n    for single in preprocessed_data:\n        white = np.nanmean(single[:, 1:], axis=1)\n        white_s = savgol_filter(white, 20, 2)\n\n        p1, p2 = _phase_detector_signal(white_s, cfg)\n        p1 = max(delta, p1)\n        p2 = min(len(white) - delta - 1, p2)\n\n        oot_left = white[: p1 - delta] if p1 - delta > 0 else np.empty(0, white.dtype)\n        oot_right = white[p2 + delta :] if (p2 + delta) < white.size else np.empty(0, white.dtype)\n        oot = np.concatenate([oot_left, oot_right]) if (oot_left.size + oot_right.size) else oot_left\n        inn = white[p1 + delta : max(p1 + delta, p2 - delta)]\n\n        if oot.size == 0 or inn.size == 0:\n            sig_rel.append(np.nan); continue\n\n        n_oot, n_in = len(oot), len(inn)\n        var_oot = np.nanvar(oot, ddof=1)\n        var_in  = np.nanvar(inn, ddof=1)\n        oot_mean = float(np.nanmean(oot)) if np.isfinite(np.nanmean(oot)) else float(np.nanmean(white))\n\n        sigma_rel = np.sqrt(var_oot / max(n_oot,1) + var_in / max(n_in,1)) / max(oot_mean, eps)\n        sig_rel.append(sigma_rel)\n\n    s = np.asarray(sig_rel, dtype=float)\n    mask = np.isfinite(s) & (s > 0)\n    med = float(np.nanmedian(s[mask])) if mask.any() else 1.0\n\n    k = np.ones_like(s)\n    if med > 0 and np.isfinite(med):\n        k[mask] = np.sqrt(s[mask] / med)\n    \n    k = np.clip(k, 0.92, 1.22)\n\n    return k * cfg.SIGMA * 1.04\n\n\nclass SignalProcessor:\n    def __init__(self, config):\n        self.cfg = config\n        self.adc_info = pd.read_csv(f\"{self.cfg.DATA_PATH}/adc_info.csv\")\n        self.planet_ids = pd.read_csv(f'{self.cfg.DATA_PATH}/{self.cfg.DATASET}_star_info.csv', index_col='planet_id').index.astype(int)\n\n    def _apply_linear_corr(self, linear_corr, signal):\n        coeffs = np.flip(linear_corr, axis=0)\n        x = signal.astype(np.float64, copy=False)\n        out = np.empty_like(x, dtype=np.float64)\n        out[...] = coeffs[0]\n        for k in range(1, coeffs.shape[0]):\n            np.multiply(out, x, out=out)\n            out += coeffs[k]\n\n        return out.astype(signal.dtype, copy=False)\n\n    def _calibrate_single_signal(self, planet_id, sensor):\n        sensor_cfg = self.cfg.SENSOR_CONFIG[sensor]\n\n        signal = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_signal_0.parquet\").to_numpy()\n        dark = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/dark.parquet\").to_numpy()\n        dead = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/dead.parquet\").to_numpy()\n        flat = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/flat.parquet\").to_numpy()\n        linear_corr = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/linear_corr.parquet\").values.astype(np.float64).reshape(sensor_cfg[\"linear_corr_shape\"])\n\n        signal = signal.reshape(sensor_cfg[\"raw_shape\"])\n        gain = self.adc_info[f\"{sensor}_adc_gain\"].iloc[0]\n        offset = self.adc_info[f\"{sensor}_adc_offset\"].iloc[0]\n        signal = signal / gain + offset\n\n        hot = sigma_clip(dark, sigma=5, maxiters=5).mask\n\n        if sensor == \"AIRS-CH0\":\n            signal = signal[:, :, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            linear_corr = linear_corr[:, :, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            dark = dark[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            dead = dead[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            flat = flat[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            hot = hot[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n\n\n        if sensor == \"FGS1\":\n            y0, y1, x0, x1 = 10, 22, 10, 22\n            signal = signal[:, y0:y1, x0:x1]\n            dark   = dark[y0:y1, x0:x1]\n            dead   = dead[y0:y1, x0:x1]\n            flat   = flat[y0:y1, x0:x1]\n            linear_corr = linear_corr[:, y0:y1, x0:x1]\n            hot    = hot[y0:y1, x0:x1]\n\n        np.maximum(signal, 0, out=signal)\n\n        if sensor == \"FGS1\":\n            signal = self._apply_linear_corr(linear_corr, signal)\n        elif sensor == \"AIRS-CH0\":\n            sl = (slice(None), slice(10, 22), slice(None))\n            signal[sl] = self._apply_linear_corr(linear_corr[:, 10:22, :], signal[sl])\n        else:\n            signal = self._apply_linear_corr(linear_corr, signal)\n\n        base_dt, increment = sensor_cfg[\"dt_pattern\"]\n        even_scale = base_dt\n        odd_scale  = base_dt + increment\n\n        signal[::2] -= dark * even_scale\n        signal[1::2] -= dark * odd_scale\n\n        \n        return signal\n\n    def _preprocess_calibrated_signal(self, calibrated_signal, sensor):\n        sensor_cfg = self.cfg.SENSOR_CONFIG[sensor]\n        binning = sensor_cfg[\"binning\"]\n\n        if sensor == \"AIRS-CH0\":\n            signal_roi = calibrated_signal[:, 10:22, :]\n        elif sensor == \"FGS1\":\n            signal_roi = calibrated_signal[:, 10:22, 10:22]\n            signal_roi = signal_roi.reshape(signal_roi.shape[0], -1)\n        \n        mean_signal = np.nanmean(signal_roi, axis=1)\n\n        cds_signal = mean_signal[1::2] - mean_signal[0::2]\n\n        n_bins = cds_signal.shape[0] // binning\n        binned = np.array([\n            cds_signal[j*binning : (j+1)*binning].mean(axis=0) \n            for j in range(n_bins)\n        ])\n\n        if sensor == \"AIRS-CH0\":\n            q_lo = np.nanpercentile(binned, 5.0, axis=1, keepdims=True)\n            q_hi = np.nanpercentile(binned, 95.0, axis=1, keepdims=True)\n            np.clip(binned, q_lo, q_hi, out=binned)\n\n        if sensor == \"FGS1\":\n            binned = binned.reshape((binned.shape[0], 1))\n\n        if sensor == \"AIRS-CH0\":\n            var = np.nanvar(binned, axis=0, ddof=1)\n            med = np.nanmedian(var)\n\n            safe_var = np.where(~np.isfinite(var) | (var <= 0), med if (np.isfinite(med) and med > 0) else 1.0, var)\n            w = 1.0 / safe_var\n\n            lo, hi = np.nanpercentile(w, 5.0), np.nanpercentile(w, 95.0)\n            if np.isfinite(lo) and np.isfinite(hi) and lo < hi:\n                w = np.clip(w, lo, hi)\n\n            M = binned.shape[1]\n            s = np.nansum(w)\n            if np.isfinite(s) and s > 0:\n                w = w * (M / s)\n            else:\n                w = np.ones_like(w)\n\n            binned *= w[None, :]\n\n\n        return binned\n\n    def _process_planet_sensor(self, args):\n        planet_id, sensor = args['planet_id'], args['sensor']\n        calibrated = self._calibrate_single_signal(planet_id, sensor)\n        preprocessed = self._preprocess_calibrated_signal(calibrated, sensor)\n        return preprocessed\n\n    def process_all_data(self):\n        args_fgs1 = [dict(planet_id=planet_id, sensor=\"FGS1\") for planet_id in self.planet_ids]\n        preprocessed_fgs1 = pqdm(args_fgs1, self._process_planet_sensor, n_jobs=self.cfg.N_JOBS)\n\n        args_airs_ch0 = [dict(planet_id=planet_id, sensor=\"AIRS-CH0\") for planet_id in self.planet_ids]\n        preprocessed_airs_ch0 = pqdm(args_airs_ch0, self._process_planet_sensor, n_jobs=self.cfg.N_JOBS)\n\n        preprocessed_signal = np.concatenate(\n            [np.stack(preprocessed_fgs1), np.stack(preprocessed_airs_ch0)], axis=2\n        )\n        \n        return preprocessed_signal\n    \n\nclass TransitModel:\n    def __init__(self, config):\n        self.cfg = config\n\n    def _phase_detector(self, signal):\n        search_slice = self.cfg.MODEL_PHASE_DETECTION_SLICE\n        min_index = np.argmin(signal[search_slice]) + search_slice.start\n        \n        signal1 = signal[:min_index]\n        signal2 = signal[min_index:]\n\n        grad1 = np.gradient(signal1)\n        grad1 /= grad1.max()\n        \n        grad2 = np.gradient(signal2)\n        grad2 /= grad2.max()\n\n        phase1 = np.argmin(grad1)\n        phase2 = np.argmax(grad2) + min_index\n\n        return phase1, phase2\n    \n    def _objective_function(self, s, signal, phase1, phase2):\n        delta = self.cfg.MODEL_OPTIMIZATION_DELTA\n        power = self.cfg.MODEL_POLYNOMIAL_DEGREE\n\n        if phase1 - delta <= 0 or phase2 + delta >= len(signal) or phase2 - delta - (phase1 + delta) < 5:\n            delta = 2\n\n        y = np.concatenate([\n            signal[: phase1 - delta],\n            signal[phase1 + delta : phase2 - delta] * (1 + s),\n            signal[phase2 + delta :]\n        ])\n        x = np.arange(len(y))\n\n        coeffs = np.polyfit(x, y, deg=power)\n        poly = np.poly1d(coeffs)\n        error = np.abs(poly(x) - y).mean()\n        \n        return error\n\n    def predict(self, single_preprocessed_signal):\n        signal_1d = single_preprocessed_signal[:, 1:].mean(axis=1)\n        signal_1d = savgol_filter(signal_1d, 20, 2)\n        \n        phase1, phase2 = self._phase_detector(signal_1d)\n\n        phase1 = max(self.cfg.MODEL_OPTIMIZATION_DELTA, phase1)\n        phase2 = min(len(signal_1d) - self.cfg.MODEL_OPTIMIZATION_DELTA - 1, phase2)    \n\n        result = minimize(\n            fun=self._objective_function,\n            x0=[0.0001],\n            args=(signal_1d, phase1, phase2),\n            method=\"Nelder-Mead\"\n        )\n        \n        return result.x[0]\n\n    def predict_all(self, preprocessed_signals):\n        predictions = [\n            self.predict(preprocessed_signal)\n            for preprocessed_signal in tqdm(preprocessed_signals)\n        ]\n        return np.array(predictions) * self.cfg.SCALE\n    \nclass SubmissionGenerator:\n    def __init__(self, config):\n        self.cfg = config\n        self.sample_submission = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2025/sample_submission.csv\", index_col=\"planet_id\")\n\n    def create(self, predictions1, predictions2, predictions, sigma_fgs=None, sigma_air=None):\n        planet_ids = self.sample_submission.index\n        n_mu = self.sample_submission.shape[1] // 2\n\n        preds = np.asarray(predictions, dtype=float).reshape(-1)\n        mu = np.tile(preds.reshape(-1, 1), (1, n_mu))\n        mu = np.clip(mu, 0, None)\n\n        sigmas = np.full_like(mu, self.cfg.SIGMA, dtype=float)\n        if sigma_fgs is not None:\n            sigma_fgs = np.asarray(sigma_fgs, dtype=float).reshape(-1)\n            sigmas[:, 0] = np.clip(sigma_fgs, 1e-6, 0.1)\n        if sigma_air is not None:\n            sigma_air = np.asarray(sigma_air, dtype=float).reshape(-1, 1)\n            sigmas[:, 1:] = np.clip(sigma_air, 1e-6, 0.1)\n\n        submission_df = pd.DataFrame(\n            np.concatenate([mu, sigmas], axis=1),\n            columns=self.sample_submission.columns,\n            index=planet_ids\n        )\n        submission_df.iloc[:, 1:283] = predictions2\n        submission_df.iloc[:, 0] = predictions1\n\n        submission_df.to_csv(\"submission_0.csv\")\n        return submission_df\n\n\n\nconfig = Config()\n    \nsignal_processor = SignalProcessor(config)\npreprocessed_data = signal_processor.process_all_data()\n\nmodel = TransitModel(config)\npredictions = model.predict_all(preprocessed_data)","metadata":{"execution":{"iopub.status.busy":"2025-09-24T05:28:25.239238Z","iopub.execute_input":"2025-09-24T05:28:25.239559Z","iopub.status.idle":"2025-09-24T05:28:31.152826Z","shell.execute_reply.started":"2025-09-24T05:28:25.239538Z","shell.execute_reply":"2025-09-24T05:28:31.151927Z"},"papermill":{"duration":5.564138,"end_time":"2025-09-19T07:12:01.438184","exception":false,"start_time":"2025-09-19T07:11:55.874046","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"processor = SignalProcessor(Config)\nStarInfo = pd.read_csv(ROOT_PATH + f\"/{MODE}_star_info.csv\")\nStarInfo[\"planet_id\"] = StarInfo[\"planet_id\"].astype(int)\nPlanetIds = StarInfo[\"planet_id\"].tolist()\nStarInfo = StarInfo.set_index(\"planet_id\")\nStarInfo","metadata":{"execution":{"iopub.status.busy":"2025-09-24T05:28:31.154579Z","iopub.execute_input":"2025-09-24T05:28:31.154863Z","iopub.status.idle":"2025-09-24T05:28:31.18275Z","shell.execute_reply.started":"2025-09-24T05:28:31.154841Z","shell.execute_reply":"2025-09-24T05:28:31.181906Z"},"papermill":{"duration":0.033405,"end_time":"2025-09-19T07:12:01.476283","exception":false,"start_time":"2025-09-19T07:12:01.442878","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = pd.DataFrame(predictions)\npredictions = predictions.rename(columns={0: \"transit_depth\"})\npredictions","metadata":{"execution":{"iopub.status.busy":"2025-09-24T05:28:31.183743Z","iopub.execute_input":"2025-09-24T05:28:31.184007Z","iopub.status.idle":"2025-09-24T05:28:31.191084Z","shell.execute_reply.started":"2025-09-24T05:28:31.183988Z","shell.execute_reply":"2025-09-24T05:28:31.190296Z"},"papermill":{"duration":0.012145,"end_time":"2025-09-19T07:12:01.492559","exception":false,"start_time":"2025-09-19T07:12:01.480414","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"input_df = StarInfo.copy()\npred_series = predictions[\"transit_depth\"].set_axis(input_df.index, copy=False)\n\ninput_df.insert(0, \"transit_depth\", (pred_series * 10000).to_numpy())\n\nfeatures = [\"transit_depth\", \"Rs\", \"i\"]\nX = input_df[features].values.astype(\"float32\")","metadata":{"execution":{"iopub.status.busy":"2025-09-24T05:28:31.191973Z","iopub.execute_input":"2025-09-24T05:28:31.192262Z","iopub.status.idle":"2025-09-24T05:28:31.208772Z","shell.execute_reply.started":"2025-09-24T05:28:31.192244Z","shell.execute_reply":"2025-09-24T05:28:31.208188Z"},"papermill":{"duration":0.012284,"end_time":"2025-09-19T07:12:01.50855","exception":false,"start_time":"2025-09-19T07:12:01.496266","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X","metadata":{"execution":{"iopub.status.busy":"2025-09-24T05:28:31.209677Z","iopub.execute_input":"2025-09-24T05:28:31.20992Z","iopub.status.idle":"2025-09-24T05:28:31.221567Z","shell.execute_reply.started":"2025-09-24T05:28:31.209898Z","shell.execute_reply":"2025-09-24T05:28:31.220992Z"},"papermill":{"duration":0.009487,"end_time":"2025-09-19T07:12:01.521671","exception":false,"start_time":"2025-09-19T07:12:01.512184","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ResidualBlock(nn.Module):\n    def __init__(self, dim, p=0.2):\n        super().__init__()\n        self.fc1 = nn.Linear(dim, dim)\n        self.bn1 = nn.BatchNorm1d(dim)\n        self.fc2 = nn.Linear(dim, dim)\n        self.bn2 = nn.BatchNorm1d(dim)\n        self.relu = nn.ReLU()\n        self.dropout = nn.Dropout(p)\n\n    def forward(self, x):\n        identity = x\n        out = self.relu(self.bn1(self.fc1(x)))\n        out = self.dropout(out)\n        out = self.bn2(self.fc2(out))\n        return self.relu(out + identity)\n\n\nclass ResNetMLP(nn.Module):\n    def __init__(self, input_dim=3, hidden_dim=32, output_dim=1, num_blocks=3, dropout_rate=0.2):\n        super().__init__()\n        self.input_layer = nn.Linear(input_dim, hidden_dim)\n        self.blocks = nn.Sequential(*[ResidualBlock(hidden_dim, p=dropout_rate) for _ in range(num_blocks)])\n        self.output_layer = nn.Linear(hidden_dim, output_dim)\n\n    def forward(self, x):\n        x = self.input_layer(x)\n        x = self.blocks(x)\n        x = self.output_layer(x)\n        return x\nmodel = ResNetMLP(num_blocks=80, dropout_rate=0.2)\nmodel.load_state_dict(torch.load(\"/kaggle/input/fgs1/pytorch/default/1/best_model.pth\"))\nmodel.eval()\nX_tensor = torch.tensor(X, dtype=torch.float32)\nwith torch.no_grad():\n    predictions1 = model(X_tensor).numpy()\npredictions1 /= 10000","metadata":{"papermill":{"duration":0.266015,"end_time":"2025-09-19T07:12:01.791349","exception":false,"start_time":"2025-09-19T07:12:01.525334","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T05:28:31.222228Z","iopub.execute_input":"2025-09-24T05:28:31.22245Z","iopub.status.idle":"2025-09-24T05:28:31.506165Z","shell.execute_reply.started":"2025-09-24T05:28:31.222425Z","shell.execute_reply":"2025-09-24T05:28:31.505569Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions1","metadata":{"execution":{"iopub.status.busy":"2025-09-24T05:28:31.506863Z","iopub.execute_input":"2025-09-24T05:28:31.507098Z","iopub.status.idle":"2025-09-24T05:28:31.512015Z","shell.execute_reply.started":"2025-09-24T05:28:31.507075Z","shell.execute_reply":"2025-09-24T05:28:31.511462Z"},"papermill":{"duration":0.010167,"end_time":"2025-09-19T07:12:01.87234","exception":false,"start_time":"2025-09-19T07:12:01.862173","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ResidualBlock2(nn.Module):\n    def __init__(self, dim, p=0.2):\n        super().__init__()\n        self.fc1 = nn.Linear(dim, dim)\n        self.bn1 = nn.BatchNorm1d(dim)\n        self.fc2 = nn.Linear(dim, dim)\n        self.bn2 = nn.BatchNorm1d(dim)\n        self.relu = nn.ReLU()\n        self.dropout = nn.Dropout(p)\n\n    def forward(self, x):\n        identity = x\n        out = self.relu(self.bn1(self.fc1(x)))\n        out = self.dropout(out)\n        out = self.bn2(self.fc2(out))\n        return self.relu(out + identity)\n\n\nclass ResNetMLP2(nn.Module):\n    def __init__(self, input_dim=3, hidden_dim=128, output_dim = 282, num_blocks=3, dropout_rate=0.2):\n        super().__init__()\n        self.input_layer = nn.Linear(input_dim, hidden_dim)\n        self.blocks = nn.Sequential(*[ResidualBlock(hidden_dim, p=dropout_rate) for _ in range(num_blocks)])\n        self.output_layer = nn.Linear(hidden_dim, output_dim)\n\n    def forward(self, x):\n        x = self.input_layer(x)\n        x = self.blocks(x)\n        x = self.output_layer(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2025-09-24T05:28:31.514705Z","iopub.execute_input":"2025-09-24T05:28:31.514924Z","iopub.status.idle":"2025-09-24T05:28:31.524749Z","shell.execute_reply.started":"2025-09-24T05:28:31.514908Z","shell.execute_reply":"2025-09-24T05:28:31.524076Z"},"papermill":{"duration":0.012091,"end_time":"2025-09-19T07:12:01.88868","exception":false,"start_time":"2025-09-19T07:12:01.876589","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model2 = ResNetMLP2(num_blocks=80, dropout_rate=0.3)\nmodel2.load_state_dict(torch.load(\"/kaggle/input/airs/pytorch/default/1/best_model_airs.pth\"))\nmodel2.eval()\n\nwith torch.no_grad():\n    predictions2 = model2(X_tensor).numpy()\npredictions2 /= 10000","metadata":{"execution":{"iopub.status.busy":"2025-09-24T05:28:31.525372Z","iopub.execute_input":"2025-09-24T05:28:31.525557Z","iopub.status.idle":"2025-09-24T05:28:31.871753Z","shell.execute_reply.started":"2025-09-24T05:28:31.525542Z","shell.execute_reply":"2025-09-24T05:28:31.870854Z"},"papermill":{"duration":0.318232,"end_time":"2025-09-19T07:12:02.211175","exception":false,"start_time":"2025-09-19T07:12:01.892943","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions2","metadata":{"execution":{"iopub.status.busy":"2025-09-24T05:28:31.872504Z","iopub.execute_input":"2025-09-24T05:28:31.872708Z","iopub.status.idle":"2025-09-24T05:28:31.879703Z","shell.execute_reply.started":"2025-09-24T05:28:31.872692Z","shell.execute_reply":"2025-09-24T05:28:31.878892Z"},"papermill":{"duration":0.012663,"end_time":"2025-09-19T07:12:02.230096","exception":false,"start_time":"2025-09-19T07:12:02.217433","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sigma_fgs_vec = estimate_sigma_fgs(preprocessed_data, config)\nsigma_air_vec = estimate_sigma_air(preprocessed_data, config)\n\n\nsubmission_generator = SubmissionGenerator(config)\nsubmission = submission_generator.create(predictions1, predictions2, predictions, sigma_fgs=sigma_fgs_vec, sigma_air=sigma_air_vec)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2025-09-24T05:28:31.880514Z","iopub.execute_input":"2025-09-24T05:28:31.881207Z","iopub.status.idle":"2025-09-24T05:28:31.926219Z","shell.execute_reply.started":"2025-09-24T05:28:31.881188Z","shell.execute_reply":"2025-09-24T05:28:31.92565Z"},"papermill":{"duration":0.032549,"end_time":"2025-09-19T07:12:02.266924","exception":false,"start_time":"2025-09-19T07:12:02.234375","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model 2","metadata":{}},{"cell_type":"code","source":"import os\nimport time\nimport itertools\nimport multiprocessing as mp\n\nimport numpy as np\nimport pandas as pd\nimport pandas.api.types\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader, TensorDataset, random_split\n\nimport matplotlib.pyplot as plt\n\nfrom tqdm import tqdm\nfrom pqdm.threads import pqdm\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import mean_squared_error\n\nfrom astropy.stats import sigma_clip\nfrom scipy.signal import savgol_filter\nfrom scipy.optimize import minimize\nimport scipy.stats\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score\nfrom sklearn.preprocessing import StandardScaler\nimport pandas as pd\nimport numpy as np\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score\n\nROOT_PATH = \"/kaggle/input/ariel-data-challenge-2025\"\nMODE = \"test\"\n\n__t0 = time.perf_counter()\n\nclass Config:\n    FEATURES = ['transit_depth', 'Rs', 'i']\n    # FEATURES = ['transit_depth', 'Rs', 'i', 'P', 'scaled_depth']\n    # FEATURES = ['transit_depth', 'Rs', 'Ms', 'Ts', 'Mp', 'e', 'P', 'sma', 'i']\n    DATA_PATH = '/kaggle/input/ariel-data-challenge-2025'\n    DATASET = \"test\"\n    load_data = False  # Set to True to load from disk, False to generate\n    DEBUG = False\n\n    SCALE = 0.96\n    SIGMA = 0.00055\n    \n    CUT_INF = 39\n    CUT_SUP = 321\n    \n    SENSOR_CONFIG = {\n        \"AIRS-CH0\": {\n            \"raw_shape\": [11250, 32, 356],\n            \"calibrated_shape\": [1, 32, CUT_SUP - CUT_INF],\n            \"linear_corr_shape\": (6, 32, 356),\n            \"dt_pattern\": (0.1, 4.5), \n            \"binning\": 30\n        },\n        \"FGS1\": {\n            \"raw_shape\": [135000, 32, 32],\n            \"calibrated_shape\": [1, 32, 32],\n            \"linear_corr_shape\": (6, 32, 32),\n            \"dt_pattern\": (0.1, 0.1),\n            \"binning\": 30 * 12\n        }\n    }\n    \n    MODEL_PHASE_DETECTION_SLICE = slice(30, 140)\n    MODEL_OPTIMIZATION_DELTA = 11 # 9\n    MODEL_POLYNOMIAL_DEGREE = 3\n    \n    N_JOBS = 3\n\nclass ParticipantVisibleError(Exception):\n    pass\n\ndef score(\n    solution: pd.DataFrame,\n    submission: pd.DataFrame,\n    row_id_column_name: str,\n    naive_mean: float,\n    naive_sigma: float,\n    fsg_sigma_true: float = 1e-6,\n    airs_sigma_true: float = 1e-5,\n    fgs_weight: float = 1,\n) -> float:\n    \"\"\"\n    This is a Gaussian Log Likelihood based metric. For a submission, which contains the predicted mean (x_hat) and variance (x_hat_std),\n    we calculate the Gaussian Log-likelihood (GLL) value to the provided ground truth (x). We treat each pair of x_hat,\n    x_hat_std as a 1D gaussian, meaning there will be 283 1D gaussian distributions, hence 283 values for each test spectrum,\n    the GLL value for one spectrum is the sum of all of them.\n\n    Inputs:\n        - solution: Ground Truth spectra (from test set)\n            - shape: (nsamples, n_wavelengths)\n        - submission: Predicted spectra and errors (from participants)\n            - shape: (nsamples, n_wavelengths*2)\n        naive_mean: (float) mean from the train set.\n        naive_sigma: (float) standard deviation from the train set.\n        fsg_sigma_true: (float) standard deviation from the FSG1 instrument for the test set.\n        airs_sigma_true: (float) standard deviation from the AIRS instrument for the test set.\n        fgs_weight: (float) relative weight of the fgs channel\n    \"\"\"\n\n    del solution[row_id_column_name]\n    del submission[row_id_column_name]\n\n    if submission.min().min() < 0:\n        raise ParticipantVisibleError('Negative values in the submission')\n    for col in submission.columns:\n        if not pandas.api.types.is_numeric_dtype(submission[col]):\n            raise ParticipantVisibleError(f'Submission column {col} must be a number')\n\n    n_wavelengths = len(solution.columns)\n    if len(submission.columns) != n_wavelengths * 2:\n        raise ParticipantVisibleError('Wrong number of columns in the submission')\n\n    y_pred = submission.iloc[:, :n_wavelengths].values\n    # Set a non-zero minimum sigma pred to prevent division by zero errors.\n    sigma_pred = np.clip(submission.iloc[:, n_wavelengths:].values, a_min=10**-15, a_max=None)\n    sigma_true = np.append(\n        np.array(\n            [\n                fsg_sigma_true,\n            ]\n        ),\n        np.ones(n_wavelengths - 1) * airs_sigma_true,\n    )\n    y_true = solution.values\n\n    GLL_pred = scipy.stats.norm.logpdf(y_true, loc=y_pred, scale=sigma_pred)\n    GLL_true = scipy.stats.norm.logpdf(y_true, loc=y_true, scale=sigma_true * np.ones_like(y_true))\n    GLL_mean = scipy.stats.norm.logpdf(y_true, loc=naive_mean * np.ones_like(y_true), scale=naive_sigma * np.ones_like(y_true))\n\n    # normalise the score, right now it becomes a matrix instead of a scalar.\n    ind_scores = (GLL_pred - GLL_mean) / (GLL_true - GLL_mean)\n\n    weights = np.append(np.array([fgs_weight]), np.ones(len(solution.columns) - 1))\n    weights = weights * np.ones_like(ind_scores)\n    submit_score = np.average(ind_scores, weights=weights)\n    return float(np.clip(submit_score, 0.0, 1.0))\n\ndef _phase_detector_signal(signal, cfg):\n    sl = cfg.MODEL_PHASE_DETECTION_SLICE\n    min_idx = int(np.argmin(signal[sl])) + sl.start\n    s1 = signal[:min_idx]; s2 = signal[min_idx:]\n    if s1.size < 3 or s2.size < 3:\n        return 0, len(signal) - 1\n    g1 = np.gradient(s1); g1_max = np.max(g1) if np.size(g1) else 0.0\n    g2 = np.gradient(s2); g2_max = np.max(g2) if np.size(g2) else 0.0\n    if g1_max != 0: g1 /= g1_max\n    if g2_max != 0: g2 /= g2_max\n    phase1 = int(np.argmin(g1)); phase2 = int(np.argmax(g2)) + min_idx\n    return phase1, phase2\n\ndef estimate_sigma_fgs(preprocessed_data, cfg):\n    \"\"\"Возвращает вектор sigma_1 (для FGS1) длиной N_planets — мягкий множитель к cfg.SIGMA.\"\"\"\n    sig_rel = []\n    delta = cfg.MODEL_OPTIMIZATION_DELTA\n    eps = 1e-12\n    for single in preprocessed_data:\n        # фазы по AIRS белой кривой — так же, как в модели\n        air_white = savgol_filter(single[:, 1:].mean(axis=1), 20, 2)\n        p1, p2 = _phase_detector_signal(air_white, cfg)\n        p1 = max(delta, p1)\n        p2 = min(len(air_white) - delta - 1, p2)\n\n        fgs = single[:, 0]\n        oot = (fgs[: p1 - delta] if p1 - delta > 0 else np.empty(0, fgs.dtype))\n        if p2 + delta < fgs.size:\n            oot = np.concatenate([oot, fgs[p2 + delta :]])\n        inn = fgs[p1 + delta : max(p1 + delta, p2 - delta)]\n\n        if oot.size == 0 or inn.size == 0:\n            sig_rel.append(np.nan); continue\n\n        n_oot, n_in = len(oot), len(inn)\n        var_oot = np.nanvar(oot, ddof=1)\n        var_in  = np.nanvar(inn, ddof=1)\n        oot_mean = float(np.nanmean(oot)) if np.isfinite(np.nanmean(oot)) else float(np.nanmean(fgs))\n        # относительная неопределённость глубины (в тех же ед., что s)\n        sigma_rel = np.sqrt(var_oot / max(n_oot,1) + var_in / max(n_in,1)) / max(oot_mean, eps)\n        sig_rel.append(sigma_rel)\n\n    s = np.asarray(sig_rel, dtype=float)\n    mask = np.isfinite(s) & (s > 0)\n    med = float(np.nanmedian(s[mask])) if mask.any() else 1.0\n\n    # мягкий множитель: корень, и узкий клип, чтобы не рисковать\n    k = np.ones_like(s)\n    if med > 0 and np.isfinite(med):\n        k[mask] = np.sqrt(s[mask] / med)\n    k = np.clip(k, 0.8, 1.25)  # ±20–25% от базовой σ\n\n    return k * cfg.SIGMA\n\ndef estimate_sigma_air(preprocessed_data, cfg):\n    \"\"\"Возвращает вектор sigma_air длиной N_planets — мягкий множитель к cfg.SIGMA для всех AIRS-каналов.\"\"\"\n    sig_rel = []\n    delta = cfg.MODEL_OPTIMIZATION_DELTA\n    eps = 1e-12\n\n    for single in preprocessed_data:\n        # белая кривая AIRS на бинированных данных (после всех твоих весов по λ)\n        white = np.nanmean(single[:, 1:], axis=1)         # (n_bins,)\n        white_s = savgol_filter(white, 20, 2)             # для фаз\n\n        p1, p2 = _phase_detector_signal(white_s, cfg)\n        p1 = max(delta, p1)\n        p2 = min(len(white) - delta - 1, p2)\n\n        oot_left = white[: p1 - delta] if p1 - delta > 0 else np.empty(0, white.dtype)\n        oot_right = white[p2 + delta :] if (p2 + delta) < white.size else np.empty(0, white.dtype)\n        oot = np.concatenate([oot_left, oot_right]) if (oot_left.size + oot_right.size) else oot_left\n        inn = white[p1 + delta : max(p1 + delta, p2 - delta)]\n\n        if oot.size == 0 or inn.size == 0:\n            sig_rel.append(np.nan); continue\n\n        n_oot, n_in = len(oot), len(inn)\n        var_oot = np.nanvar(oot, ddof=1)\n        var_in  = np.nanvar(inn, ddof=1)\n        oot_mean = float(np.nanmean(oot)) if np.isfinite(np.nanmean(oot)) else float(np.nanmean(white))\n\n        sigma_rel = np.sqrt(var_oot / max(n_oot,1) + var_in / max(n_in,1)) / max(oot_mean, eps)\n        sig_rel.append(sigma_rel)\n\n    s = np.asarray(sig_rel, dtype=float)\n    mask = np.isfinite(s) & (s > 0)\n    med = float(np.nanmedian(s[mask])) if mask.any() else 1.0\n\n    # мягкий множитель вокруг медианы\n    k = np.ones_like(s)\n    if med > 0 and np.isfinite(med):\n        k[mask] = np.sqrt(s[mask] / med)\n    k = np.clip(k, 0.90, 1.20)  # ±10%–20%\n\n    return k * cfg.SIGMA\n\nclass SignalProcessor:\n    def __init__(self, config):\n        self.cfg = config\n        self.adc_info = pd.read_csv(f\"{self.cfg.DATA_PATH}/adc_info.csv\")\n        self.planet_ids = pd.read_csv(f'{self.cfg.DATA_PATH}/{self.cfg.DATASET}_star_info.csv', index_col='planet_id').index.astype(int)\n\n    def _apply_linear_corr(self, linear_corr, signal):\n\n        coeffs = np.flip(linear_corr, axis=0)      # shape: (D, X, Y), D — старшая степень сначала\n        x = signal.astype(np.float64, copy=False)  # считаем в float64 для стабильности\n        out = np.empty_like(x, dtype=np.float64)\n        out[...] = coeffs[0]  # broadcast (X,Y) -> (T,X,Y)\n        for k in range(1, coeffs.shape[0]):\n            np.multiply(out, x, out=out)  # in-place умножение\n            out += coeffs[k]              # broadcast (X,Y)\n\n        return out.astype(signal.dtype, copy=False)\n\n    def _calibrate_single_signal(self, planet_id, sensor):\n        \"\"\"\n        Калибровка single-node сигнала.\n        Политика масок: DEAD — маскируем, HOT — НЕ маскируем (оставляем в данных).\n        \"\"\"\n        sensor_cfg = self.cfg.SENSOR_CONFIG[sensor]\n    \n        # --- load ---\n        signal = pd.read_parquet(\n            f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_signal_0.parquet\"\n        ).to_numpy()\n        dark = pd.read_parquet(\n            f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/dark.parquet\"\n        ).to_numpy()\n        dead = pd.read_parquet(\n            f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/dead.parquet\"\n        ).to_numpy()\n        flat = pd.read_parquet(\n            f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/flat.parquet\"\n        ).to_numpy()\n        linear_corr = pd.read_parquet(\n            f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/linear_corr.parquet\"\n        ).values.astype(np.float64).reshape(sensor_cfg[\"linear_corr_shape\"])\n    \n        # --- reshape & ADC ---\n        signal = signal.reshape(sensor_cfg[\"raw_shape\"])\n        gain = self.adc_info[f\"{sensor}_adc_gain\"].iloc[0]\n        offset = self.adc_info[f\"{sensor}_adc_offset\"].iloc[0]\n        signal = signal / gain + offset  # сохраняем твою формулу\n    \n        # HOT только для мониторинга, не для маскирования\n        hot = sigma_clip(dark, sigma=5, maxiters=5).mask\n    \n        # --- crop per sensor ---\n        if sensor == \"AIRS-CH0\":\n            signal = signal[:, :, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            linear_corr = linear_corr[:, :, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            dark = dark[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            dead = dead[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            flat = flat[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            hot = hot[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]  # только для логов\n    \n        if sensor == \"FGS1\":\n            y0, y1, x0, x1 = 10, 22, 10, 22\n            signal = signal[:, y0:y1, x0:x1]\n            dark   = dark[y0:y1, x0:x1]\n            dead   = dead[y0:y1, x0:x1]\n            flat   = flat[y0:y1, x0:x1]\n            linear_corr = linear_corr[:, y0:y1, x0:x1]\n            hot    = hot[y0:y1, x0:x1]  # только для логов\n    \n        # --- non-neg clamp before linearity corr (как у тебя) ---\n        np.maximum(signal, 0, out=signal)\n    \n        # --- linearity correction ---\n        if sensor == \"FGS1\":\n            signal = self._apply_linear_corr(linear_corr, signal)\n        elif sensor == \"AIRS-CH0\":\n            sl = (slice(None), slice(10, 22), slice(None))  # T, Y, λ\n            signal[sl] = self._apply_linear_corr(linear_corr[:, 10:22, :], signal[sl])\n        else:\n            signal = self._apply_linear_corr(linear_corr, signal)\n    \n        # --- dark subtraction с учётом паттерна интеграций ---\n        base_dt, increment = sensor_cfg[\"dt_pattern\"]\n        even_scale = base_dt\n        odd_scale  = base_dt + increment\n        signal[::2]  -= dark * even_scale\n        signal[1::2] -= dark * odd_scale\n    \n        # --- APPLY FLAT (HOT-KEEP: не включаем hot в маску!) ---\n        if sensor == \"FGS1\":\n            flat_roi = flat.astype(signal.dtype, copy=False).copy()      # (12,12)\n            bad = (dead) | ~np.isfinite(flat_roi) | (flat_roi == 0)      # ← ТОЛЬКО dead/invalid\n            flat_roi[bad] = np.nan\n            signal /= flat_roi\n    \n        elif sensor == \"AIRS-CH0\":\n            y0, y1 = 10, 22\n            flat_roi = flat[y0:y1, :].astype(signal.dtype, copy=False).copy()  # (12, λ)\n            bad = (dead[y0:y1, :]) | ~np.isfinite(flat_roi) | (flat_roi == 0)  # ← ТОЛЬКО dead/invalid\n            flat_roi[bad] = np.nan\n            signal[:, y0:y1, :] /= flat_roi\n    \n        else:\n            flat2 = flat.astype(signal.dtype, copy=False).copy()\n            bad2 = (dead) | ~np.isfinite(flat2) | (flat2 == 0)                  # ← ТОЛЬКО dead/invalid\n            flat2[bad2] = np.nan\n            signal /= flat2\n        # --- END FLAT ---\n    \n        # (опционально) логируем метрики hot/dead\n        if getattr(self.cfg, \"LOG_HOT_STATS\", False):\n            if not hasattr(self, \"stats\"):\n                self.stats = []\n            self.stats.append({\n                \"planet_id\": int(planet_id),\n                \"sensor\": sensor,\n                \"hot_frac\": float(np.mean(hot)),\n                \"dead_frac\": float(np.mean(dead)),\n            })\n    \n        return signal\n\n    def _preprocess_calibrated_signal(self, calibrated_signal, sensor):\n        sensor_cfg = self.cfg.SENSOR_CONFIG[sensor]\n        binning = sensor_cfg[\"binning\"]\n\n        if sensor == \"AIRS-CH0\":\n            signal_roi = calibrated_signal[:, 10:22, :]\n        elif sensor == \"FGS1\":\n            signal_roi = calibrated_signal[:, 10:22, 10:22]\n            signal_roi = signal_roi.reshape(signal_roi.shape[0], -1)\n        \n        mean_signal = np.nanmean(signal_roi, axis=1)\n\n        cds_signal = mean_signal[1::2] - mean_signal[0::2]\n\n        n_bins = cds_signal.shape[0] // binning\n        binned = np.array([\n            cds_signal[j*binning : (j+1)*binning].mean(axis=0) \n            for j in range(n_bins)\n        ])\n\n        if sensor == \"AIRS-CH0\":\n            q_lo = np.nanpercentile(binned, 5.0, axis=1, keepdims=True)    # (n_bins, 1)\n            q_hi = np.nanpercentile(binned, 95.0, axis=1, keepdims=True)   # (n_bins, 1)\n            np.clip(binned, q_lo, q_hi, out=binned)\n\n        if sensor == \"FGS1\":\n            binned = binned.reshape((binned.shape[0], 1))\n\n        if sensor == \"AIRS-CH0\":\n            var = np.nanvar(binned, axis=0, ddof=1)                 # (λ, )\n            med = np.nanmedian(var)\n            safe_var = np.where(~np.isfinite(var) | (var <= 0), med if (np.isfinite(med) and med > 0) else 1.0, var)\n            w = 1.0 / safe_var\n\n            lo, hi = np.nanpercentile(w, 5.0), np.nanpercentile(w, 95.0)\n            if np.isfinite(lo) and np.isfinite(hi) and lo < hi:\n                w = np.clip(w, lo, hi)\n\n            M = binned.shape[1]\n            s = np.nansum(w)\n            if np.isfinite(s) and s > 0:\n                w = w * (M / s)\n            else:\n                w = np.ones_like(w)\n\n            binned *= w[None, :]\n\n\n        return binned\n\n    def _process_planet_sensor(self, args):\n        planet_id, sensor = args['planet_id'], args['sensor']\n        calibrated = self._calibrate_single_signal(planet_id, sensor)\n        preprocessed = self._preprocess_calibrated_signal(calibrated, sensor)\n        return preprocessed\n\n    def process_all_data(self):\n        args_fgs1 = [dict(planet_id=planet_id, sensor=\"FGS1\") for planet_id in self.planet_ids]\n        preprocessed_fgs1 = pqdm(args_fgs1, self._process_planet_sensor, n_jobs=self.cfg.N_JOBS)\n\n        args_airs_ch0 = [dict(planet_id=planet_id, sensor=\"AIRS-CH0\") for planet_id in self.planet_ids]\n        preprocessed_airs_ch0 = pqdm(args_airs_ch0, self._process_planet_sensor, n_jobs=self.cfg.N_JOBS)\n\n        preprocessed_signal = np.concatenate(\n            [np.stack(preprocessed_fgs1), np.stack(preprocessed_airs_ch0)], axis=2\n        )\n        return preprocessed_signal\n    \n\nclass TransitModel:\n    def __init__(self, config):\n        self.cfg = config\n\n    def _phase_detector(self, signal):\n        search_slice = self.cfg.MODEL_PHASE_DETECTION_SLICE\n        min_index = np.argmin(signal[search_slice]) + search_slice.start\n        \n        signal1 = signal[:min_index]\n        signal2 = signal[min_index:]\n\n        grad1 = np.gradient(signal1)\n        grad1 /= grad1.max()\n        \n        grad2 = np.gradient(signal2)\n        grad2 /= grad2.max()\n\n        phase1 = np.argmin(grad1)\n        phase2 = np.argmax(grad2) + min_index\n\n        return phase1, phase2\n    \n    def _objective_function(self, s, signal, phase1, phase2):\n        delta = self.cfg.MODEL_OPTIMIZATION_DELTA\n        power = self.cfg.MODEL_POLYNOMIAL_DEGREE\n\n        if phase1 - delta <= 0 or phase2 + delta >= len(signal) or phase2 - delta - (phase1 + delta) < 5:\n            delta = 2\n\n        y = np.concatenate([\n            signal[: phase1 - delta],\n            signal[phase1 + delta : phase2 - delta] * (1 + s),\n            signal[phase2 + delta :]\n        ])\n        x = np.arange(len(y))\n\n        coeffs = np.polyfit(x, y, deg=power)\n        poly = np.poly1d(coeffs)\n        error = np.abs(poly(x) - y).mean()\n        \n        return error\n\n    def predict(self, single_preprocessed_signal):\n        signal_1d = single_preprocessed_signal[:, 1:].mean(axis=1)\n        signal_1d = savgol_filter(signal_1d, 23, 2)\n        \n        phase1, phase2 = self._phase_detector(signal_1d)\n\n        phase1 = max(self.cfg.MODEL_OPTIMIZATION_DELTA, phase1)\n        phase2 = min(len(signal_1d) - self.cfg.MODEL_OPTIMIZATION_DELTA - 1, phase2)    \n\n        result = minimize(\n            fun=self._objective_function,\n            x0=[0.0001],\n            args=(signal_1d, phase1, phase2),\n            method=\"Nelder-Mead\"\n        )\n        \n        return result.x[0]\n\n    def predict_all(self, preprocessed_signals):\n        predictions = [\n            self.predict(preprocessed_signal)\n            for preprocessed_signal in tqdm(preprocessed_signals)\n        ]\n        return np.array(predictions) * self.cfg.SCALE\n\nStarInfo = pd.read_csv(ROOT_PATH + f\"/{MODE}_star_info.csv\")\nStarInfo[\"planet_id\"] = StarInfo[\"planet_id\"].astype(int)\nPlanetIds = StarInfo[\"planet_id\"].tolist()\nStarInfo = StarInfo.set_index(\"planet_id\")\nclass SubmissionGenerator:\n    def __init__(self, config):\n        self.cfg = config\n        self.sample_submission = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2025/sample_submission.csv\", index_col=\"planet_id\")\n\n    def create(self, predictions1, predictions, sigma_fgs=None, sigma_air=None):\n        planet_ids = self.sample_submission.index\n        n_mu = self.sample_submission.shape[1] // 2  # 283\n\n        preds = np.asarray(predictions, dtype=float).reshape(-1)\n        mu = np.tile(preds.reshape(-1, 1), (1, n_mu))\n        mu = np.clip(mu, 0, None)\n\n        sigmas = np.full_like(mu, self.cfg.SIGMA, dtype=float)\n        if sigma_fgs is not None:\n            sigma_fgs = np.asarray(sigma_fgs, dtype=float).reshape(-1)\n            sigmas[:, 0] = np.clip(sigma_fgs, 1e-6, 0.1)\n        if sigma_air is not None:\n            sigma_air = np.asarray(sigma_air, dtype=float).reshape(-1, 1)\n            sigmas[:, 1:] = np.clip(sigma_air, 1e-6, 0.1)\n\n        submission_df = pd.DataFrame(\n            np.concatenate([mu, sigmas], axis=1),\n            columns=self.sample_submission.columns,\n            index=planet_ids\n        )\n        submission_df.iloc[:, 0] = predictions\n        submission_df.iloc[:, 1:283] = predictions1\n        submission_df.to_csv(\"submission_1.csv\")\n        \n        return submission_df\n\nclass SEBlock(nn.Module):\n    def __init__(self, dim, reduction=16):\n        super().__init__()\n        self.fc1 = nn.Linear(dim, dim // reduction, bias=False)\n        self.fc2 = nn.Linear(dim // reduction, dim, bias=False)\n        self.act = nn.SiLU()   # Or nn.ReLU()\n\n    def forward(self, x):\n        # Compute channel-wise attention\n        w = x.mean(dim=0, keepdim=True)      # Global context (mean across batch)\n        w = self.act(self.fc1(w))\n        w = torch.sigmoid(self.fc2(w))\n        return x * w   # Rescale input\n\nclass AttentionBlock(nn.Module):\n    def __init__(self, dim, num_heads=4, dropout=0.1):\n        super().__init__()\n        self.attn = nn.MultiheadAttention(embed_dim=dim, num_heads=num_heads, batch_first=True)\n        self.norm = nn.LayerNorm(dim)\n        self.dropout = nn.Dropout(dropout)\n\n    def forward(self, x):\n        # Expect shape [batch, seq_len, dim], so expand if only [batch, dim]\n        if x.dim() == 2:\n            x = x.unsqueeze(1)   # -> [batch, 1, dim]\n        attn_out, _ = self.attn(x, x, x)\n        out = self.norm(x + self.dropout(attn_out))\n        return out.squeeze(1)    # Back to [batch, dim]\n\nclass ResidualBlock2(nn.Module):\n    def __init__(self, dim, p=0.2):\n        super().__init__()\n        self.fc1 = nn.Linear(dim, dim)\n        self.fc2 = nn.Linear(dim, dim)\n        self.relu = nn.ReLU()\n        self.dropout = nn.Dropout(p)\n\n    def forward(self, x):\n        identity = x\n        out = self.relu(self.fc1(x))\n        out = self.dropout(out)\n        out = self.fc2(out)\n        return self.relu(out + identity)\n\nclass ResNetMLP2(nn.Module):\n    def __init__(self, input_dim=3, hidden_dim=128, output_dim=282, num_blocks=3, dropout_rate=0.2):\n        super().__init__()\n        self.input_layer = nn.Linear(input_dim, hidden_dim)\n        self.blocks = nn.Sequential(*[ResidualBlock2(hidden_dim, p=dropout_rate) for _ in range(num_blocks)])\n        self.output_layer = nn.Linear(hidden_dim, output_dim)\n\n    def forward(self, x):\n        x = self.input_layer(x)\n        x = self.blocks(x)\n        x = self.output_layer(x)\n        return x\n\n\ndef load_cv_models_and_scalers(directory):\n    \"\"\"\n    Loads all cross-validation models and scalers from the specified directory.\n    Args:\n        directory (str): Path to the directory containing model and scaler files.\n    Returns:\n        all_models (list): List of loaded models.\n        scaler_X: Loaded X scaler.\n        scaler_y: Loaded y scaler.\n    \"\"\"\n    import os\n    import joblib\n    # Load scalers\n    scaler_X = joblib.load(os.path.join(directory, 'scaler_X.joblib'))\n    scaler_y = joblib.load(os.path.join(directory, 'scaler_y.joblib'))\n\n    # Load all CV models\n    all_models = []\n    model_params = {\n        'input_dim': len(Config.FEATURES),  # Always use the current feature count\n        'hidden_dim': 256,\n        'output_dim': 282,\n        'num_blocks': 35,\n        'dropout_rate': 0.1\n    }\n    for fold in range(1, 11):\n        model = ResNetMLP2(**model_params).double()\n        model_path = os.path.join(directory, f'best_model_airs_cv_fold{fold}.pth')\n        model.load_state_dict(torch.load(model_path, map_location=torch.device('cpu')))\n        model.eval()\n        all_models.append(model)\n    return all_models, scaler_X, scaler_y\n    \nall_models, scaler_X, scaler_y = load_cv_models_and_scalers('/kaggle/input/arial-dataset-new-version/ariel-2025-result/results_v37')\n\nconfig = Config()\nsignal_processor = SignalProcessor(config)\npreprocessed_data = signal_processor.process_all_data()\n\nmodel = TransitModel(config)\npredictions = model.predict_all(preprocessed_data)\nsigma_fgs_vec = estimate_sigma_fgs(preprocessed_data, config)\nsigma_air_vec = estimate_sigma_air(preprocessed_data, config)\n\npredictions_df = pd.DataFrame({\n    \"planet_id\": PlanetIds,\n    \"transit_depth\": predictions\n})\n\ninput_df = pd.merge(predictions_df, StarInfo, on=\"planet_id\", how=\"left\")\ninput_df['scaled_depth'] = input_df['transit_depth'] / input_df['Rs']\n\nX = input_df[Config.FEATURES].values.astype(np.float64)\nX_scaled = scaler_X.transform(X)\nX_tensor = torch.tensor(X_scaled, dtype=torch.float64)\n\n# Generate average prediction from all CV models (on scaled X, then inverse transform)\nwith torch.no_grad():\n    preds_scaled = [model(X_tensor).numpy() for model in all_models]\npredictions1_scaled = np.mean(preds_scaled, axis=0)\npredictions1 = scaler_y.inverse_transform(predictions1_scaled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T05:28:31.927027Z","iopub.execute_input":"2025-09-24T05:28:31.927264Z","iopub.status.idle":"2025-09-24T05:28:41.620079Z","shell.execute_reply.started":"2025-09-24T05:28:31.927236Z","shell.execute_reply":"2025-09-24T05:28:41.619394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    FEATURES = ['transit_depth', 'Rs', 'i', 'P']\n\n    DATA_PATH = '/kaggle/input/ariel-data-challenge-2025'\n    DATASET = \"test\"\n    load_data = False  # Set to True to load from disk, False to generate\n    DEBUG = False\n\n    SCALE = 0.96\n    SIGMA = 0.00055\n    \n    CUT_INF = 39\n    CUT_SUP = 321\n    \n    SENSOR_CONFIG = {\n        \"AIRS-CH0\": {\n            \"raw_shape\": [11250, 32, 356],\n            \"calibrated_shape\": [1, 32, CUT_SUP - CUT_INF],\n            \"linear_corr_shape\": (6, 32, 356),\n            \"dt_pattern\": (0.1, 4.5), \n            \"binning\": 30\n        },\n        \"FGS1\": {\n            \"raw_shape\": [135000, 32, 32],\n            \"calibrated_shape\": [1, 32, 32],\n            \"linear_corr_shape\": (6, 32, 32),\n            \"dt_pattern\": (0.1, 0.1),\n            \"binning\": 30 * 12\n        }\n    }\n    \n    MODEL_PHASE_DETECTION_SLICE = slice(30, 140)\n    MODEL_OPTIMIZATION_DELTA = 11 # 9\n    MODEL_POLYNOMIAL_DEGREE = 3\n    \n    N_JOBS = 3\n\n\ndef load_cv_models_and_scalers(directory):\n    \"\"\"\n    Loads all cross-validation models and scalers from the specified directory.\n    Args:\n        directory (str): Path to the directory containing model and scaler files.\n    Returns:\n        all_models (list): List of loaded models.\n        scaler_X: Loaded X scaler.\n        scaler_y: Loaded y scaler.\n    \"\"\"\n    import os\n    import joblib\n    # Load scalers\n    scaler_X = joblib.load(os.path.join(directory, 'scaler_X.joblib'))\n    scaler_y = joblib.load(os.path.join(directory, 'scaler_y.joblib'))\n\n    # Load all CV models\n    all_models = []\n    model_params = {\n        'input_dim': len(Config.FEATURES),  # Always use the current feature count\n        'hidden_dim': 256,\n        'output_dim': 282,\n        'num_blocks': 35,\n        'dropout_rate': 0.1\n    }\n    for fold in range(1, 11):\n        model = ResNetMLP2(**model_params).double()\n        model_path = os.path.join(directory, f'best_model_airs_cv_fold{fold}.pth')\n        model.load_state_dict(torch.load(model_path, map_location=torch.device('cpu')))\n        model.eval()\n        all_models.append(model)\n    return all_models, scaler_X, scaler_y\n    \nall_models, scaler_X, scaler_y = load_cv_models_and_scalers('/kaggle/input/arial-dataset-new-version/ariel-2025-result/results_sv44')\nconfig = Config()\nsignal_processor = SignalProcessor(config)\npreprocessed_data = signal_processor.process_all_data()\n\nmodel = TransitModel(config)\npredictions = model.predict_all(preprocessed_data)\nsigma_fgs_vec = estimate_sigma_fgs(preprocessed_data, config)\nsigma_air_vec = estimate_sigma_air(preprocessed_data, config)\n\npredictions_df = pd.DataFrame({\n    \"planet_id\": PlanetIds,\n    \"transit_depth\": predictions\n})\n\ninput_df = pd.merge(predictions_df, StarInfo, on=\"planet_id\", how=\"left\")\nX = input_df[Config.FEATURES].values.astype(np.float64)\nX_scaled = scaler_X.transform(X)\nX_tensor = torch.tensor(X_scaled, dtype=torch.float64)\n\n# Generate average prediction from all CV models (on scaled X, then inverse transform)\nwith torch.no_grad():\n    preds_scaled = [model(X_tensor).numpy() for model in all_models]\n\n# Correctly average the predictions\npredictions2_scaled = np.mean(preds_scaled, axis=0)\n\n# Correctly inverse transform the *averaged* predictions\npredictions2 = scaler_y.inverse_transform(predictions2_scaled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T05:28:41.620883Z","iopub.execute_input":"2025-09-24T05:28:41.62115Z","iopub.status.idle":"2025-09-24T05:28:50.434535Z","shell.execute_reply.started":"2025-09-24T05:28:41.621127Z","shell.execute_reply":"2025-09-24T05:28:50.433862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Combine the predictions into a single array\nall_predictions = np.array([predictions1, predictions2])\n\n# Calculate the mean across the models (axis=0)\nfinal_predictions = np.mean(all_predictions, axis=0)\n\nsubmission_generator = SubmissionGenerator(config)\nsubmission = submission_generator.create(final_predictions, predictions, sigma_fgs=sigma_fgs_vec, sigma_air=sigma_air_vec)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T05:28:50.435308Z","iopub.execute_input":"2025-09-24T05:28:50.435581Z","iopub.status.idle":"2025-09-24T05:28:50.453889Z","shell.execute_reply.started":"2025-09-24T05:28:50.435556Z","shell.execute_reply":"2025-09-24T05:28:50.453316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb = pd.read_csv(\"submission_1.csv\") # 370\nnn =  pd.read_csv(\"submission_0.csv\") # 374","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T05:28:50.454785Z","iopub.execute_input":"2025-09-24T05:28:50.455074Z","iopub.status.idle":"2025-09-24T05:28:50.482766Z","shell.execute_reply.started":"2025-09-24T05:28:50.455048Z","shell.execute_reply":"2025-09-24T05:28:50.482262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n# Identify columns\nwl_cols    = [c for c in xgb.columns if c.startswith(\"wl_\")]\nsigma_cols = [c for c in xgb.columns if c.startswith(\"sigma_\")]\n\n# Weighted blends\nwl_blend    = 0.2 * xgb[wl_cols] + 0.8 * nn[wl_cols]\nsigma_blend = 0.4 * xgb[sigma_cols] + 0.6 * nn[sigma_cols]\n\n# Combine with planet_id\nsubmission = pd.concat(\n    [xgb[[\"planet_id\"]], wl_blend, sigma_blend],\n    axis=1\n)\n\n# Save final submission\nsubmission.to_csv(\"submission_380.csv\", index=False)\nprint(\"Saved submission.csv with shape:\", submission.shape)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T05:28:50.483472Z","iopub.execute_input":"2025-09-24T05:28:50.483719Z","iopub.status.idle":"2025-09-24T05:28:50.495866Z","shell.execute_reply.started":"2025-09-24T05:28:50.483702Z","shell.execute_reply":"2025-09-24T05:28:50.495132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T05:28:50.496516Z","iopub.execute_input":"2025-09-24T05:28:50.49672Z","iopub.status.idle":"2025-09-24T05:28:50.51739Z","shell.execute_reply.started":"2025-09-24T05:28:50.496705Z","shell.execute_reply":"2025-09-24T05:28:50.516831Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# series 2","metadata":{}},{"cell_type":"code","source":"import os\nimport time\nimport itertools\nimport multiprocessing as mp\n\nimport numpy as np\nimport pandas as pd\nimport pandas.api.types\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader, TensorDataset, random_split\n\nimport matplotlib.pyplot as plt\n\nfrom tqdm import tqdm\nfrom pqdm.threads import pqdm\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import mean_squared_error\n\nfrom astropy.stats import sigma_clip\nfrom scipy.signal import savgol_filter\nfrom scipy.optimize import minimize\nimport scipy.stats\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score\nfrom sklearn.preprocessing import StandardScaler\nimport pandas as pd\nimport numpy as np\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score\n\nROOT_PATH = \"/kaggle/input/ariel-data-challenge-2025\"\nMODE = \"test\"\n\n__t0 = time.perf_counter()\n\nclass Config:\n    FEATURES = ['transit_depth', 'Rs', 'i']\n    # FEATURES = ['transit_depth', 'Rs', 'Ms', 'Ts', 'Mp', 'e', 'P', 'sma', 'i']\n    DATA_PATH = '/kaggle/input/ariel-data-challenge-2025'\n    DATASET = \"test\"\n    load_data = True  # Set to True to load from disk, False to generate\n    DEBUG = False\n\n    SCALE = 0.96\n    SIGMA = 0.00055\n    \n    CUT_INF = 39\n    CUT_SUP = 321\n    \n    SENSOR_CONFIG = {\n        \"AIRS-CH0\": {\n            \"raw_shape\": [11250, 32, 356],\n            \"calibrated_shape\": [1, 32, CUT_SUP - CUT_INF],\n            \"linear_corr_shape\": (6, 32, 356),\n            \"dt_pattern\": (0.1, 4.5), \n            \"binning\": 30\n        },\n        \"FGS1\": {\n            \"raw_shape\": [135000, 32, 32],\n            \"calibrated_shape\": [1, 32, 32],\n            \"linear_corr_shape\": (6, 32, 32),\n            \"dt_pattern\": (0.1, 0.1),\n            \"binning\": 30 * 12\n        }\n    }\n    \n    MODEL_PHASE_DETECTION_SLICE = slice(30, 140)\n    MODEL_OPTIMIZATION_DELTA = 11 # 9\n    MODEL_POLYNOMIAL_DEGREE = 3\n    \n    N_JOBS = 3\n\nclass ParticipantVisibleError(Exception):\n    pass\n\ndef score(\n    solution: pd.DataFrame,\n    submission: pd.DataFrame,\n    row_id_column_name: str,\n    naive_mean: float,\n    naive_sigma: float,\n    fsg_sigma_true: float = 1e-6,\n    airs_sigma_true: float = 1e-5,\n    fgs_weight: float = 1,\n) -> float:\n    \"\"\"\n    This is a Gaussian Log Likelihood based metric. For a submission, which contains the predicted mean (x_hat) and variance (x_hat_std),\n    we calculate the Gaussian Log-likelihood (GLL) value to the provided ground truth (x). We treat each pair of x_hat,\n    x_hat_std as a 1D gaussian, meaning there will be 283 1D gaussian distributions, hence 283 values for each test spectrum,\n    the GLL value for one spectrum is the sum of all of them.\n\n    Inputs:\n        - solution: Ground Truth spectra (from test set)\n            - shape: (nsamples, n_wavelengths)\n        - submission: Predicted spectra and errors (from participants)\n            - shape: (nsamples, n_wavelengths*2)\n        naive_mean: (float) mean from the train set.\n        naive_sigma: (float) standard deviation from the train set.\n        fsg_sigma_true: (float) standard deviation from the FSG1 instrument for the test set.\n        airs_sigma_true: (float) standard deviation from the AIRS instrument for the test set.\n        fgs_weight: (float) relative weight of the fgs channel\n    \"\"\"\n\n    del solution[row_id_column_name]\n    del submission[row_id_column_name]\n\n    if submission.min().min() < 0:\n        raise ParticipantVisibleError('Negative values in the submission')\n    for col in submission.columns:\n        if not pandas.api.types.is_numeric_dtype(submission[col]):\n            raise ParticipantVisibleError(f'Submission column {col} must be a number')\n\n    n_wavelengths = len(solution.columns)\n    if len(submission.columns) != n_wavelengths * 2:\n        raise ParticipantVisibleError('Wrong number of columns in the submission')\n\n    y_pred = submission.iloc[:, :n_wavelengths].values\n    # Set a non-zero minimum sigma pred to prevent division by zero errors.\n    sigma_pred = np.clip(submission.iloc[:, n_wavelengths:].values, a_min=10**-15, a_max=None)\n    sigma_true = np.append(\n        np.array(\n            [\n                fsg_sigma_true,\n            ]\n        ),\n        np.ones(n_wavelengths - 1) * airs_sigma_true,\n    )\n    y_true = solution.values\n\n    GLL_pred = scipy.stats.norm.logpdf(y_true, loc=y_pred, scale=sigma_pred)\n    GLL_true = scipy.stats.norm.logpdf(y_true, loc=y_true, scale=sigma_true * np.ones_like(y_true))\n    GLL_mean = scipy.stats.norm.logpdf(y_true, loc=naive_mean * np.ones_like(y_true), scale=naive_sigma * np.ones_like(y_true))\n\n    # normalise the score, right now it becomes a matrix instead of a scalar.\n    ind_scores = (GLL_pred - GLL_mean) / (GLL_true - GLL_mean)\n\n    weights = np.append(np.array([fgs_weight]), np.ones(len(solution.columns) - 1))\n    weights = weights * np.ones_like(ind_scores)\n    submit_score = np.average(ind_scores, weights=weights)\n    return float(np.clip(submit_score, 0.0, 1.0))\n\ndef _phase_detector_signal(signal, cfg):\n    sl = cfg.MODEL_PHASE_DETECTION_SLICE\n    min_idx = int(np.argmin(signal[sl])) + sl.start\n    s1 = signal[:min_idx]; s2 = signal[min_idx:]\n    if s1.size < 3 or s2.size < 3:\n        return 0, len(signal) - 1\n    g1 = np.gradient(s1); g1_max = np.max(g1) if np.size(g1) else 0.0\n    g2 = np.gradient(s2); g2_max = np.max(g2) if np.size(g2) else 0.0\n    if g1_max != 0: g1 /= g1_max\n    if g2_max != 0: g2 /= g2_max\n    phase1 = int(np.argmin(g1)); phase2 = int(np.argmax(g2)) + min_idx\n    return phase1, phase2\n\ndef estimate_sigma_fgs(preprocessed_data, cfg):\n    \"\"\"Возвращает вектор sigma_1 (для FGS1) длиной N_planets — мягкий множитель к cfg.SIGMA.\"\"\"\n    sig_rel = []\n    delta = cfg.MODEL_OPTIMIZATION_DELTA\n    eps = 1e-12\n    for single in preprocessed_data:\n        # фазы по AIRS белой кривой — так же, как в модели\n        air_white = savgol_filter(single[:, 1:].mean(axis=1), 20, 2)\n        p1, p2 = _phase_detector_signal(air_white, cfg)\n        p1 = max(delta, p1)\n        p2 = min(len(air_white) - delta - 1, p2)\n\n        fgs = single[:, 0]\n        oot = (fgs[: p1 - delta] if p1 - delta > 0 else np.empty(0, fgs.dtype))\n        if p2 + delta < fgs.size:\n            oot = np.concatenate([oot, fgs[p2 + delta :]])\n        inn = fgs[p1 + delta : max(p1 + delta, p2 - delta)]\n\n        if oot.size == 0 or inn.size == 0:\n            sig_rel.append(np.nan); continue\n\n        n_oot, n_in = len(oot), len(inn)\n        var_oot = np.nanvar(oot, ddof=1)\n        var_in  = np.nanvar(inn, ddof=1)\n        oot_mean = float(np.nanmean(oot)) if np.isfinite(np.nanmean(oot)) else float(np.nanmean(fgs))\n        # относительная неопределённость глубины (в тех же ед., что s)\n        sigma_rel = np.sqrt(var_oot / max(n_oot,1) + var_in / max(n_in,1)) / max(oot_mean, eps)\n        sig_rel.append(sigma_rel)\n\n    s = np.asarray(sig_rel, dtype=float)\n    mask = np.isfinite(s) & (s > 0)\n    med = float(np.nanmedian(s[mask])) if mask.any() else 1.0\n\n    # мягкий множитель: корень, и узкий клип, чтобы не рисковать\n    k = np.ones_like(s)\n    if med > 0 and np.isfinite(med):\n        k[mask] = np.sqrt(s[mask] / med)\n    k = np.clip(k, 0.8, 1.25)  # ±20–25% от базовой σ\n\n    return k * cfg.SIGMA\n\ndef estimate_sigma_air(preprocessed_data, cfg):\n    \"\"\"Возвращает вектор sigma_air длиной N_planets — мягкий множитель к cfg.SIGMA для всех AIRS-каналов.\"\"\"\n    sig_rel = []\n    delta = cfg.MODEL_OPTIMIZATION_DELTA\n    eps = 1e-12\n\n    for single in preprocessed_data:\n        # белая кривая AIRS на бинированных данных (после всех твоих весов по λ)\n        white = np.nanmean(single[:, 1:], axis=1)         # (n_bins,)\n        white_s = savgol_filter(white, 20, 2)             # для фаз\n\n        p1, p2 = _phase_detector_signal(white_s, cfg)\n        p1 = max(delta, p1)\n        p2 = min(len(white) - delta - 1, p2)\n\n        oot_left = white[: p1 - delta] if p1 - delta > 0 else np.empty(0, white.dtype)\n        oot_right = white[p2 + delta :] if (p2 + delta) < white.size else np.empty(0, white.dtype)\n        oot = np.concatenate([oot_left, oot_right]) if (oot_left.size + oot_right.size) else oot_left\n        inn = white[p1 + delta : max(p1 + delta, p2 - delta)]\n\n        if oot.size == 0 or inn.size == 0:\n            sig_rel.append(np.nan); continue\n\n        n_oot, n_in = len(oot), len(inn)\n        var_oot = np.nanvar(oot, ddof=1)\n        var_in  = np.nanvar(inn, ddof=1)\n        oot_mean = float(np.nanmean(oot)) if np.isfinite(np.nanmean(oot)) else float(np.nanmean(white))\n\n        sigma_rel = np.sqrt(var_oot / max(n_oot,1) + var_in / max(n_in,1)) / max(oot_mean, eps)\n        sig_rel.append(sigma_rel)\n\n    s = np.asarray(sig_rel, dtype=float)\n    mask = np.isfinite(s) & (s > 0)\n    med = float(np.nanmedian(s[mask])) if mask.any() else 1.0\n\n    # мягкий множитель вокруг медианы\n    k = np.ones_like(s)\n    if med > 0 and np.isfinite(med):\n        k[mask] = np.sqrt(s[mask] / med)\n    k = np.clip(k, 0.90, 1.20)  # ±10%–20%\n\n    return k * cfg.SIGMA\n\nclass SignalProcessor:\n    def __init__(self, config):\n        self.cfg = config\n        self.adc_info = pd.read_csv(f\"{self.cfg.DATA_PATH}/adc_info.csv\")\n        self.planet_ids = pd.read_csv(f'{self.cfg.DATA_PATH}/{self.cfg.DATASET}_star_info.csv', index_col='planet_id').index.astype(int)\n\n    def _apply_linear_corr(self, linear_corr, signal):\n\n        coeffs = np.flip(linear_corr, axis=0)      # shape: (D, X, Y), D — старшая степень сначала\n        x = signal.astype(np.float64, copy=False)  # считаем в float64 для стабильности\n        out = np.empty_like(x, dtype=np.float64)\n        out[...] = coeffs[0]  # broadcast (X,Y) -> (T,X,Y)\n        for k in range(1, coeffs.shape[0]):\n            np.multiply(out, x, out=out)  # in-place умножение\n            out += coeffs[k]              # broadcast (X,Y)\n\n        return out.astype(signal.dtype, copy=False)\n\n    def _calibrate_single_signal(self, planet_id, sensor):\n        \"\"\"\n        Калибровка single-node сигнала.\n        Политика масок: DEAD — маскируем, HOT — НЕ маскируем (оставляем в данных).\n        \"\"\"\n        sensor_cfg = self.cfg.SENSOR_CONFIG[sensor]\n    \n        # --- load ---\n        signal = pd.read_parquet(\n            f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_signal_0.parquet\"\n        ).to_numpy()\n        dark = pd.read_parquet(\n            f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/dark.parquet\"\n        ).to_numpy()\n        dead = pd.read_parquet(\n            f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/dead.parquet\"\n        ).to_numpy()\n        flat = pd.read_parquet(\n            f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/flat.parquet\"\n        ).to_numpy()\n        linear_corr = pd.read_parquet(\n            f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/linear_corr.parquet\"\n        ).values.astype(np.float64).reshape(sensor_cfg[\"linear_corr_shape\"])\n    \n        # --- reshape & ADC ---\n        signal = signal.reshape(sensor_cfg[\"raw_shape\"])\n        gain = self.adc_info[f\"{sensor}_adc_gain\"].iloc[0]\n        offset = self.adc_info[f\"{sensor}_adc_offset\"].iloc[0]\n        signal = signal / gain + offset  # сохраняем твою формулу\n    \n        # HOT только для мониторинга, не для маскирования\n        hot = sigma_clip(dark, sigma=5, maxiters=5).mask\n    \n        # --- crop per sensor ---\n        if sensor == \"AIRS-CH0\":\n            signal = signal[:, :, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            linear_corr = linear_corr[:, :, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            dark = dark[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            dead = dead[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            flat = flat[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            hot = hot[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]  # только для логов\n    \n        if sensor == \"FGS1\":\n            y0, y1, x0, x1 = 10, 22, 10, 22\n            signal = signal[:, y0:y1, x0:x1]\n            dark   = dark[y0:y1, x0:x1]\n            dead   = dead[y0:y1, x0:x1]\n            flat   = flat[y0:y1, x0:x1]\n            linear_corr = linear_corr[:, y0:y1, x0:x1]\n            hot    = hot[y0:y1, x0:x1]  # только для логов\n    \n        # --- non-neg clamp before linearity corr (как у тебя) ---\n        np.maximum(signal, 0, out=signal)\n    \n        # --- linearity correction ---\n        if sensor == \"FGS1\":\n            signal = self._apply_linear_corr(linear_corr, signal)\n        elif sensor == \"AIRS-CH0\":\n            sl = (slice(None), slice(10, 22), slice(None))  # T, Y, λ\n            signal[sl] = self._apply_linear_corr(linear_corr[:, 10:22, :], signal[sl])\n        else:\n            signal = self._apply_linear_corr(linear_corr, signal)\n    \n        # --- dark subtraction с учётом паттерна интеграций ---\n        base_dt, increment = sensor_cfg[\"dt_pattern\"]\n        even_scale = base_dt\n        odd_scale  = base_dt + increment\n        signal[::2]  -= dark * even_scale\n        signal[1::2] -= dark * odd_scale\n    \n        # --- APPLY FLAT (HOT-KEEP: не включаем hot в маску!) ---\n        if sensor == \"FGS1\":\n            flat_roi = flat.astype(signal.dtype, copy=False).copy()      # (12,12)\n            bad = (dead) | ~np.isfinite(flat_roi) | (flat_roi == 0)      # ← ТОЛЬКО dead/invalid\n            flat_roi[bad] = np.nan\n            signal /= flat_roi\n    \n        elif sensor == \"AIRS-CH0\":\n            y0, y1 = 10, 22\n            flat_roi = flat[y0:y1, :].astype(signal.dtype, copy=False).copy()  # (12, λ)\n            bad = (dead[y0:y1, :]) | ~np.isfinite(flat_roi) | (flat_roi == 0)  # ← ТОЛЬКО dead/invalid\n            flat_roi[bad] = np.nan\n            signal[:, y0:y1, :] /= flat_roi\n    \n        else:\n            flat2 = flat.astype(signal.dtype, copy=False).copy()\n            bad2 = (dead) | ~np.isfinite(flat2) | (flat2 == 0)                  # ← ТОЛЬКО dead/invalid\n            flat2[bad2] = np.nan\n            signal /= flat2\n        # --- END FLAT ---\n    \n        # (опционально) логируем метрики hot/dead\n        if getattr(self.cfg, \"LOG_HOT_STATS\", False):\n            if not hasattr(self, \"stats\"):\n                self.stats = []\n            self.stats.append({\n                \"planet_id\": int(planet_id),\n                \"sensor\": sensor,\n                \"hot_frac\": float(np.mean(hot)),\n                \"dead_frac\": float(np.mean(dead)),\n            })\n    \n        return signal\n\n    def _preprocess_calibrated_signal(self, calibrated_signal, sensor):\n        sensor_cfg = self.cfg.SENSOR_CONFIG[sensor]\n        binning = sensor_cfg[\"binning\"]\n\n        if sensor == \"AIRS-CH0\":\n            signal_roi = calibrated_signal[:, 10:22, :]\n        elif sensor == \"FGS1\":\n            signal_roi = calibrated_signal[:, 10:22, 10:22]\n            signal_roi = signal_roi.reshape(signal_roi.shape[0], -1)\n        \n        mean_signal = np.nanmean(signal_roi, axis=1)\n\n        cds_signal = mean_signal[1::2] - mean_signal[0::2]\n\n        n_bins = cds_signal.shape[0] // binning\n        binned = np.array([\n            cds_signal[j*binning : (j+1)*binning].mean(axis=0) \n            for j in range(n_bins)\n        ])\n\n        if sensor == \"AIRS-CH0\":\n            q_lo = np.nanpercentile(binned, 5.0, axis=1, keepdims=True)    # (n_bins, 1)\n            q_hi = np.nanpercentile(binned, 95.0, axis=1, keepdims=True)   # (n_bins, 1)\n            np.clip(binned, q_lo, q_hi, out=binned)\n\n        if sensor == \"FGS1\":\n            binned = binned.reshape((binned.shape[0], 1))\n\n        if sensor == \"AIRS-CH0\":\n            var = np.nanvar(binned, axis=0, ddof=1)                 # (λ, )\n            med = np.nanmedian(var)\n            safe_var = np.where(~np.isfinite(var) | (var <= 0), med if (np.isfinite(med) and med > 0) else 1.0, var)\n            w = 1.0 / safe_var\n\n            lo, hi = np.nanpercentile(w, 5.0), np.nanpercentile(w, 95.0)\n            if np.isfinite(lo) and np.isfinite(hi) and lo < hi:\n                w = np.clip(w, lo, hi)\n\n            M = binned.shape[1]\n            s = np.nansum(w)\n            if np.isfinite(s) and s > 0:\n                w = w * (M / s)\n            else:\n                w = np.ones_like(w)\n\n            binned *= w[None, :]\n\n\n        return binned\n\n    def _process_planet_sensor(self, args):\n        planet_id, sensor = args['planet_id'], args['sensor']\n        calibrated = self._calibrate_single_signal(planet_id, sensor)\n        preprocessed = self._preprocess_calibrated_signal(calibrated, sensor)\n        return preprocessed\n\n    def process_all_data(self):\n        args_fgs1 = [dict(planet_id=planet_id, sensor=\"FGS1\") for planet_id in self.planet_ids]\n        preprocessed_fgs1 = pqdm(args_fgs1, self._process_planet_sensor, n_jobs=self.cfg.N_JOBS)\n\n        args_airs_ch0 = [dict(planet_id=planet_id, sensor=\"AIRS-CH0\") for planet_id in self.planet_ids]\n        preprocessed_airs_ch0 = pqdm(args_airs_ch0, self._process_planet_sensor, n_jobs=self.cfg.N_JOBS)\n\n        preprocessed_signal = np.concatenate(\n            [np.stack(preprocessed_fgs1), np.stack(preprocessed_airs_ch0)], axis=2\n        )\n        return preprocessed_signal\n    \n\nclass TransitModel:\n    def __init__(self, config):\n        self.cfg = config\n\n    def _phase_detector(self, signal):\n        search_slice = self.cfg.MODEL_PHASE_DETECTION_SLICE\n        min_index = np.argmin(signal[search_slice]) + search_slice.start\n        \n        signal1 = signal[:min_index]\n        signal2 = signal[min_index:]\n\n        grad1 = np.gradient(signal1)\n        grad1 /= grad1.max()\n        \n        grad2 = np.gradient(signal2)\n        grad2 /= grad2.max()\n\n        phase1 = np.argmin(grad1)\n        phase2 = np.argmax(grad2) + min_index\n\n        return phase1, phase2\n    \n    def _objective_function(self, s, signal, phase1, phase2):\n        delta = self.cfg.MODEL_OPTIMIZATION_DELTA\n        power = self.cfg.MODEL_POLYNOMIAL_DEGREE\n\n        if phase1 - delta <= 0 or phase2 + delta >= len(signal) or phase2 - delta - (phase1 + delta) < 5:\n            delta = 2\n\n        y = np.concatenate([\n            signal[: phase1 - delta],\n            signal[phase1 + delta : phase2 - delta] * (1 + s),\n            signal[phase2 + delta :]\n        ])\n        x = np.arange(len(y))\n\n        coeffs = np.polyfit(x, y, deg=power)\n        poly = np.poly1d(coeffs)\n        error = np.abs(poly(x) - y).mean()\n        \n        return error\n\n    def predict(self, single_preprocessed_signal):\n        signal_1d = single_preprocessed_signal[:, 1:].mean(axis=1)\n        signal_1d = savgol_filter(signal_1d, 23, 2)\n        \n        phase1, phase2 = self._phase_detector(signal_1d)\n\n        phase1 = max(self.cfg.MODEL_OPTIMIZATION_DELTA, phase1)\n        phase2 = min(len(signal_1d) - self.cfg.MODEL_OPTIMIZATION_DELTA - 1, phase2)    \n\n        result = minimize(\n            fun=self._objective_function,\n            x0=[0.0001],\n            args=(signal_1d, phase1, phase2),\n            method=\"Nelder-Mead\"\n        )\n        \n        return result.x[0]\n\n    def predict_all(self, preprocessed_signals):\n        predictions = [\n            self.predict(preprocessed_signal)\n            for preprocessed_signal in tqdm(preprocessed_signals)\n        ]\n        return np.array(predictions) * self.cfg.SCALE\n\nStarInfo = pd.read_csv(ROOT_PATH + f\"/{MODE}_star_info.csv\")\nStarInfo[\"planet_id\"] = StarInfo[\"planet_id\"].astype(int)\nPlanetIds = StarInfo[\"planet_id\"].tolist()\nStarInfo = StarInfo.set_index(\"planet_id\")\nclass SubmissionGenerator:\n    def __init__(self, config):\n        self.cfg = config\n        self.sample_submission = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2025/sample_submission.csv\", index_col=\"planet_id\")\n\n    def create(self, predictions1, predictions, sigma_fgs=None, sigma_air=None):\n        planet_ids = self.sample_submission.index\n        n_mu = self.sample_submission.shape[1] // 2  # 283\n\n        preds = np.asarray(predictions, dtype=float).reshape(-1)\n        mu = np.tile(preds.reshape(-1, 1), (1, n_mu))\n        mu = np.clip(mu, 0, None)\n\n        sigmas = np.full_like(mu, self.cfg.SIGMA, dtype=float)\n        if sigma_fgs is not None:\n            sigma_fgs = np.asarray(sigma_fgs, dtype=float).reshape(-1)\n            sigmas[:, 0] = np.clip(sigma_fgs, 1e-6, 0.1)\n        if sigma_air is not None:\n            sigma_air = np.asarray(sigma_air, dtype=float).reshape(-1, 1)\n            sigmas[:, 1:] = np.clip(sigma_air, 1e-6, 0.1)\n\n        submission_df = pd.DataFrame(\n            np.concatenate([mu, sigmas], axis=1),\n            columns=self.sample_submission.columns,\n            index=planet_ids\n        )\n        submission_df.iloc[:, 0] = predictions\n        submission_df.iloc[:, 1:283] = predictions1\n        submission_df.to_csv(\"submission.csv\")\n        \n        return submission_df\n\nclass ResidualBlock2(nn.Module):\n    def __init__(self, dim, p=0.2):\n        super().__init__()\n        self.fc1 = nn.Linear(dim, dim)\n        self.fc2 = nn.Linear(dim, dim)\n        self.relu = nn.ReLU()\n        self.dropout = nn.Dropout(p)\n\n    def forward(self, x):\n        identity = x\n        out = self.relu(self.fc1(x))\n        out = self.dropout(out)\n        out = self.fc2(out)\n        return self.relu(out + identity)\n\nclass ResNetMLP2(nn.Module):\n    def __init__(self, input_dim=3, hidden_dim=128, output_dim=282, num_blocks=3, dropout_rate=0.2):\n        super().__init__()\n        self.input_layer = nn.Linear(input_dim, hidden_dim)\n        self.blocks = nn.Sequential(*[ResidualBlock2(hidden_dim, p=dropout_rate) for _ in range(num_blocks)])\n        self.output_layer = nn.Linear(hidden_dim, output_dim)\n\n    def forward(self, x):\n        x = self.input_layer(x)\n        x = self.blocks(x)\n        x = self.output_layer(x)\n        return x\n\n# Execute the training\n# Train and get scalers\n# all_models, scaler_X, scaler_y = train_new_model()\n\ndef load_cv_models_and_scalers(directory):\n    \"\"\"\n    Loads all cross-validation models and scalers from the specified directory.\n    Args:\n        directory (str): Path to the directory containing model and scaler files.\n    Returns:\n        all_models (list): List of loaded models.\n        scaler_X: Loaded X scaler.\n        scaler_y: Loaded y scaler.\n    \"\"\"\n    import os\n    import joblib\n    # Load scalers\n    scaler_X = joblib.load(os.path.join(directory, 'scaler_X.joblib'))\n    scaler_y = joblib.load(os.path.join(directory, 'scaler_y.joblib'))\n\n    # Load all CV models\n    all_models = []\n    model_params = {\n        'input_dim': len(Config.FEATURES),  # Always use the current feature count\n        'hidden_dim': 256,\n        'output_dim': 282,\n        'num_blocks': 80,\n        'dropout_rate': 0.3\n    }\n    for fold in range(1, 11):\n        model = ResNetMLP2(**model_params).double()\n        model_path = os.path.join(directory, f'best_model_airs_cv_fold{fold}.pth')\n        model.load_state_dict(torch.load(model_path, map_location=torch.device('cpu')))\n        model.eval()\n        all_models.append(model)\n    return all_models, scaler_X, scaler_y\n    \nall_models, scaler_X, scaler_y = load_cv_models_and_scalers('/kaggle/input/ariel-2025-result/results_3_v25')\nconfig = Config()\nsignal_processor = SignalProcessor(config)\npreprocessed_data = signal_processor.process_all_data()\n\nmodel = TransitModel(config)\npredictions = model.predict_all(preprocessed_data)\nsigma_fgs_vec = estimate_sigma_fgs(preprocessed_data, config)\nsigma_air_vec = estimate_sigma_air(preprocessed_data, config)\n\npredictions_df = pd.DataFrame({\n    \"planet_id\": PlanetIds,\n    \"transit_depth\": predictions\n})\n\ninput_df = pd.merge(predictions_df, StarInfo, on=\"planet_id\", how=\"left\")\nX = input_df[Config.FEATURES].values.astype(np.float64)\nX_scaled = scaler_X.transform(X)\nX_tensor = torch.tensor(X_scaled, dtype=torch.float64)\n\n# Generate average prediction from all CV models (on scaled X, then inverse transform)\nwith torch.no_grad():\n    preds_scaled = [model(X_tensor).numpy() for model in all_models]\npredictions1_scaled = np.mean(preds_scaled, axis=0)\npredictions1 = scaler_y.inverse_transform(predictions1_scaled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T05:28:50.518317Z","iopub.execute_input":"2025-09-24T05:28:50.518999Z","iopub.status.idle":"2025-09-24T05:29:04.489859Z","shell.execute_reply.started":"2025-09-24T05:28:50.518967Z","shell.execute_reply":"2025-09-24T05:29:04.489187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    # FEATURES = ['transit_depth', 'Rs', 'i']\n    FEATURES = ['transit_depth', 'Rs', 'Ms', 'Ts', 'Mp', 'e', 'P', 'sma', 'i']\n    DATA_PATH = '/kaggle/input/ariel-data-challenge-2025'\n    DATASET = \"test\"\n    load_data = True  # Set to True to load from disk, False to generate\n    DEBUG = False\n\n    SCALE = 0.96\n    SIGMA = 0.00055\n    \n    CUT_INF = 39\n    CUT_SUP = 321\n    \n    SENSOR_CONFIG = {\n        \"AIRS-CH0\": {\n            \"raw_shape\": [11250, 32, 356],\n            \"calibrated_shape\": [1, 32, CUT_SUP - CUT_INF],\n            \"linear_corr_shape\": (6, 32, 356),\n            \"dt_pattern\": (0.1, 4.5), \n            \"binning\": 30\n        },\n        \"FGS1\": {\n            \"raw_shape\": [135000, 32, 32],\n            \"calibrated_shape\": [1, 32, 32],\n            \"linear_corr_shape\": (6, 32, 32),\n            \"dt_pattern\": (0.1, 0.1),\n            \"binning\": 30 * 12\n        }\n    }\n    \n    MODEL_PHASE_DETECTION_SLICE = slice(30, 140)\n    MODEL_OPTIMIZATION_DELTA = 11 # 9\n    MODEL_POLYNOMIAL_DEGREE = 3\n    \n    N_JOBS = 3\n\ndef load_cv_models_and_scalers(directory):\n    \"\"\"\n    Loads all cross-validation models and scalers from the specified directory.\n    Args:\n        directory (str): Path to the directory containing model and scaler files.\n    Returns:\n        all_models (list): List of loaded models.\n        scaler_X: Loaded X scaler.\n        scaler_y: Loaded y scaler.\n    \"\"\"\n    import os\n    import joblib\n    # Load scalers\n    scaler_X = joblib.load(os.path.join(directory, 'scaler_X.joblib'))\n    scaler_y = joblib.load(os.path.join(directory, 'scaler_y.joblib'))\n\n    # Load all CV models\n    all_models = []\n    model_params = {\n        'input_dim': len(Config.FEATURES),  # Always use the current feature count\n        'hidden_dim': 256,\n        'output_dim': 282,\n        'num_blocks': 80,\n        'dropout_rate': 0.3\n    }\n    for fold in range(1, 11):\n        model = ResNetMLP2(**model_params).double()\n        model_path = os.path.join(directory, f'best_model_airs_cv_fold{fold}.pth')\n        model.load_state_dict(torch.load(model_path, map_location=torch.device('cpu')))\n        model.eval()\n        all_models.append(model)\n    return all_models, scaler_X, scaler_y\n    \nall_models, scaler_X, scaler_y = load_cv_models_and_scalers('/kaggle/input/ariel-2025-result/results_9_v26')\nconfig = Config()\nsignal_processor = SignalProcessor(config)\npreprocessed_data = signal_processor.process_all_data()\n\nmodel = TransitModel(config)\npredictions = model.predict_all(preprocessed_data)\nsigma_fgs_vec = estimate_sigma_fgs(preprocessed_data, config)\nsigma_air_vec = estimate_sigma_air(preprocessed_data, config)\n\npredictions_df = pd.DataFrame({\n    \"planet_id\": PlanetIds,\n    \"transit_depth\": predictions\n})\n\ninput_df = pd.merge(predictions_df, StarInfo, on=\"planet_id\", how=\"left\")\nX = input_df[Config.FEATURES].values.astype(np.float64)\nX_scaled = scaler_X.transform(X)\nX_tensor = torch.tensor(X_scaled, dtype=torch.float64)\n\n# Generate average prediction from all CV models (on scaled X, then inverse transform)\nwith torch.no_grad():\n    preds_scaled = [model(X_tensor).numpy() for model in all_models]\n\n# Correctly average the predictions\npredictions2_scaled = np.mean(preds_scaled, axis=0)\n\n# Correctly inverse transform the *averaged* predictions\npredictions2 = scaler_y.inverse_transform(predictions2_scaled)\n\nall_predictions = np.array([predictions1, predictions2])\n\n# Calculate the mean across the models (axis=0)\nfinal_predictions = np.mean(all_predictions, axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T05:29:04.490682Z","iopub.execute_input":"2025-09-24T05:29:04.491128Z","iopub.status.idle":"2025-09-24T05:29:19.002254Z","shell.execute_reply.started":"2025-09-24T05:29:04.491104Z","shell.execute_reply":"2025-09-24T05:29:19.001316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_generator = SubmissionGenerator(config)\nsubmission = submission_generator.create(final_predictions, predictions, sigma_fgs=sigma_fgs_vec, sigma_air=sigma_air_vec)\n\nsubmission.to_csv(\"submission_nn.csv\")\n__t1 = time.perf_counter()\nelapsed = __t1 - __t0\nprint(f\"[TIMING] total runtime: {elapsed:.2f} s ({elapsed/60:.2f} min)\")\npd.read_csv(\"submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T05:29:19.003184Z","iopub.execute_input":"2025-09-24T05:29:19.00343Z","iopub.status.idle":"2025-09-24T05:29:19.046123Z","shell.execute_reply.started":"2025-09-24T05:29:19.003413Z","shell.execute_reply":"2025-09-24T05:29:19.045529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pickle\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nimport scipy.stats\nimport xgboost as xgb\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.multioutput import MultiOutputRegressor\nfrom sklearn.preprocessing import PolynomialFeatures\nfrom tqdm import tqdm\nfrom joblib import Parallel, delayed\nimport multiprocessing\n\n# --- Set to True for training, False for submission ---\nTRAIN = False \n# Set to True to use a small subset for faster debugging\nDEBUG = False \n\nclass Config:\n    DATA_PATH = '/kaggle/input/ariel-data-challenge-2025/'\n    OUTPUT_PATH = '/kaggle/input/5-fold-arial-xgb/ariel_kfold_models' # Save models to /kaggle/working/\n    TRAIN_LABELS_PATH = os.path.join(DATA_PATH, 'train.csv')\n    TRAIN_STAR_INFO_PATH = os.path.join(DATA_PATH, 'train_star_info.csv')\n    TEST_STAR_INFO_PATH = os.path.join(DATA_PATH, 'test_star_info.csv')\n    SAMPLE_SUBMISSION_PATH = os.path.join(DATA_PATH, 'sample_submission.csv')\n    N_FOLDS = 5\n    RANDOM_STATE = 42\n    \n    XGB_PARAMS_MEAN = {\n        'objective': 'reg:squarederror',\n        'n_estimators': 1200, # Slightly more estimators for smaller fold data\n        'learning_rate': 0.02,\n        'max_depth': 7,\n        'subsample': 0.8,\n        'colsample_bytree': 0.7,\n        'random_state': RANDOM_STATE,\n        'tree_method': 'hist',\n        'n_jobs': -1, \n    }\n    \n    XGB_PARAMS_SIGMA = {\n        'objective': 'reg:squarederror',\n        'n_estimators': 600,\n        'learning_rate': 0.025,\n        'max_depth': 6,\n        'subsample': 0.7,\n        'colsample_bytree': 0.6,\n        'random_state': RANDOM_STATE,\n        'tree_method': 'hist',\n        'n_jobs': -1,\n    }\n    \n    CALIBRATION_SCALING_FACTORS = [0.8, 0.9, 1.0, 1.1, 1.2]\n    CALIBRATION_ADDITIVE_FACTORS = [0.0001, 0.0002, 0.0003, 0.0004, 0.0005]\n\nconfig = Config()\n\n# --- All helper functions (extract_light_curve_with_aperture, process_single_planet, etc.) remain the same ---\n# (I've omitted them here for brevity, but they should be included in your script exactly as they were)\ndef extract_light_curve_with_aperture(signal_df: pl.DataFrame, instrument: str):\n    if instrument == 'FGS1':\n        IMAGE_SHAPE = (32, 32)\n        APERTURE_RADIUS = 3 \n    else: # AIRS-CH0\n        IMAGE_SHAPE = (32, 356)\n        APERTURE_HEIGHT = 4 \n\n    try:\n        images = signal_df.to_numpy(zero_copy_only=True).reshape(-1, *IMAGE_SHAPE)\n    except Exception:\n        images = signal_df.to_numpy().reshape(-1, *IMAGE_SHAPE)\n        \n    median_frame = np.median(images, axis=0)\n\n    if instrument == 'FGS1':\n        center_y, center_x = np.unravel_index(np.argmax(median_frame), median_frame.shape)\n        y_start = max(0, center_y - APERTURE_RADIUS)\n        y_end = min(IMAGE_SHAPE[0], center_y + APERTURE_RADIUS + 1)\n        x_start = max(0, center_x - APERTURE_RADIUS)\n        x_end = min(IMAGE_SHAPE[1], center_x + APERTURE_RADIUS + 1)\n        aperture_flux = images[:, y_start:y_end, x_start:x_end].sum(axis=(1, 2))\n    else: # AIRS-CH0\n        vertical_profile = median_frame.sum(axis=1)\n        center_y = np.argmax(vertical_profile)\n        y_start = max(0, center_y - APERTURE_HEIGHT)\n        y_end = min(IMAGE_SHAPE[0], center_y + APERTURE_HEIGHT + 1)\n        aperture_flux = images[:, y_start:y_end, :].sum(axis=(1, 2))\n        \n    return aperture_flux.astype(np.float32)\n\ndef process_single_planet(planet_id, dataset, instrument):\n    planet_signals = []\n    for obs_count in range(2):\n        path = f'/kaggle/input/ariel-data-challenge-2025/{dataset}/{int(planet_id)}/{instrument}_signal_{obs_count}.parquet'\n        if os.path.exists(path):\n            signal_df = pl.read_parquet(path)\n            aperture_flux = extract_light_curve_with_aperture(signal_df, instrument)\n            net_signal = aperture_flux[1::2] - aperture_flux[0::2]\n            planet_signals.append(net_signal)\n    return planet_id, planet_signals\n\ndef load_all_observations(dataset, planet_ids, instrument='FGS1'):\n    print(f\"Loading ALL observations for {instrument} in {dataset} set...\")\n    results = Parallel(n_jobs=-1)(\n        delayed(process_single_planet)(pid, dataset, instrument) \n        for pid in tqdm(planet_ids, desc=f\"Dispatching {instrument} jobs\")\n    )\n    return {planet_id: signals for planet_id, signals in results if signals}\n\ndef autocorrelation(x, lag=1):\n    return np.corrcoef(x[:-lag], x[lag:])[0, 1]\n\ndef get_detrended_features(raw_data, transit_slice):\n    time_axis = np.arange(raw_data.shape[0])\n    out_of_transit_mask = np.ones_like(time_axis, dtype=bool)\n    out_of_transit_mask[transit_slice] = False\n    coeffs = np.polyfit(time_axis[out_of_transit_mask], raw_data[out_of_transit_mask], 2)\n    poly_fit = np.poly1d(coeffs)\n    trend = poly_fit(time_axis).clip(1e-6)\n    normalized_data = raw_data / trend\n    detrended_transit = normalized_data[transit_slice]\n    return {\n        'detrended_std': detrended_transit.std(), \n        'detrended_skew': scipy.stats.skew(detrended_transit),\n        'detrended_kurtosis': scipy.stats.kurtosis(detrended_transit)\n    }\n\ndef add_physics_and_interaction_features(features_df):\n    print(\"Adding physics-informed and interaction features...\")\n    df = features_df.copy()\n    transit_depth_proxy = df['fgs_depth_mean'].clip(0)\n    df['planet_star_radius_ratio'] = np.sqrt(transit_depth_proxy)\n    b = df['impact_parameter_b']\n    p = df['planet_star_radius_ratio']\n    arg1_sqrt = ((1 + p)**2 - b**2).clip(0)\n    arg1 = (np.sqrt(arg1_sqrt) / df['sma']).clip(-1, 1)\n    df['transit_duration_est'] = (df['P'] / np.pi) * np.arcsin(arg1)\n    arg2_sqrt = ((1 - p)**2 - b**2).clip(0)\n    arg2 = (np.sqrt(arg2_sqrt) / df['sma']).clip(-1, 1)\n    T_flat_est = (df['P'] / np.pi) * np.arcsin(arg2)\n    df['ingress_egress_duration_est'] = (df['transit_duration_est'] - T_flat_est) / 2.0\n    df['ingress_duration_ratio'] = (df['ingress_egress_duration_est'] / df['transit_duration_est']).fillna(0)\n    df['planet_eq_temp_est'] = df['Ts'] * np.sqrt(1 / (2 * df['sma']))\n    df['planet_gravity_proxy'] = df['Mp'] / (p * df['Rs'])**2\n    df['eclipsed_light_proxy'] = df['fgs_depth_mean'] * (df['Rs']**2)\n    df['duration_period_ratio'] = df['transit_duration_est'] / df['P']\n    print(f\"Added {8} new physics-based features.\")\n    return df\n\ndef extract_timeseries_features(f_raw, a_raw):\n    features = {}\n    fgs_transit_slice = slice(23500, 44000)\n    fgs_out_of_transit_mask = np.ones(len(f_raw), dtype=bool); fgs_out_of_transit_mask[fgs_transit_slice] = False\n    fgs_unobscured_mean = np.mean(f_raw[fgs_out_of_transit_mask])\n    fgs_transit = f_raw[fgs_transit_slice]\n    fgs_depth = (fgs_unobscured_mean - np.mean(fgs_transit)) / fgs_unobscured_mean\n    features['fgs_depth'] = fgs_depth\n    for i in range(5):\n        features[f'fgs_slice_{i+1}'] = (fgs_unobscured_mean - np.mean(fgs_transit[i*4100:(i+1)*4100])) / fgs_unobscured_mean\n    features['fgs_transit_std'] = fgs_transit.std()\n    features['fgs_snr'] = fgs_depth / (fgs_transit.std() + 1e-6)\n    features.update({f'fgs_{k}': v for k, v in get_detrended_features(f_raw, fgs_transit_slice).items()})\n    features['fgs_noise_autocorr'] = autocorrelation(f_raw[fgs_out_of_transit_mask])\n    airs_transit_slice = slice(1950, 3700)\n    airs_out_of_transit_mask = np.ones(len(a_raw), dtype=bool); airs_out_of_transit_mask[airs_transit_slice] = False\n    airs_unobscured_mean = np.mean(a_raw[airs_out_of_transit_mask])\n    airs_transit = a_raw[airs_transit_slice]\n    airs_depth = (airs_unobscured_mean - np.mean(airs_transit)) / airs_unobscured_mean\n    features['airs_depth'] = airs_depth\n    slice_len = len(airs_transit) // 5\n    for i in range(5):\n        features[f'airs_slice_{i+1}'] = (airs_unobscured_mean - np.mean(airs_transit[i*slice_len:(i+1)*slice_len])) / airs_unobscured_mean\n    features['airs_transit_std'] = airs_transit.std()\n    features['airs_snr'] = airs_depth / (airs_transit.std() + 1e-6)\n    features.update({f'airs_{k}': v for k, v in get_detrended_features(a_raw, airs_transit_slice).items()})\n    features['airs_noise_autocorr'] = autocorrelation(a_raw[airs_out_of_transit_mask])\n    features['depth_ratio_fgs_airs'] = fgs_depth / (airs_depth + 1e-6)\n    return features\n\ndef combined_feature_engineering(fgs_signals, airs_signals, star_info_df):\n    print(\"Engineering and Aggregating features...\")\n    all_planet_features_list = []\n    for planet_id in tqdm(star_info_df.index, desc=\"Feature Engineering\"):\n        num_obs = len(fgs_signals.get(planet_id, []))\n        if num_obs == 0: continue\n        obs_features_list = [extract_timeseries_features(fgs_signals[planet_id][i], airs_signals[planet_id][i]) for i in range(num_obs)]\n        obs_features_df = pd.DataFrame(obs_features_list)\n        aggregated_feats = {f'{col}_{agg}': obs_features_df[col].agg(agg) for col in obs_features_df.columns for agg in ['mean', 'std', 'min', 'max']}\n        aggregated_feats['planet_id'] = planet_id\n        aggregated_feats['num_observations'] = num_obs\n        all_planet_features_list.append(aggregated_feats)\n    \n    features_df = pd.DataFrame(all_planet_features_list).set_index('planet_id')\n    meta_df = star_info_df.copy().fillna(star_info_df.median())\n    meta_df['impact_parameter_b'] = meta_df['sma'] * np.cos(np.deg2rad(meta_df['i']))\n    meta_df['rho_star_proxy'] = meta_df['Ms'] / (meta_df['Rs']**3)\n    \n    poly = PolynomialFeatures(degree=2, include_bias=False, interaction_only=True)\n    poly_cols = ['Rs', 'Ts', 'Mp', 'P', 'impact_parameter_b', 'rho_star_proxy']\n    poly_features = poly.fit_transform(meta_df[poly_cols])\n    poly_df = pd.DataFrame(poly_features, columns=poly.get_feature_names_out(poly_cols), index=meta_df.index)\n    \n    final_features_df = pd.concat([features_df, meta_df, poly_df], axis=1).loc[:, ~pd.concat([features_df, meta_df, poly_df], axis=1).columns.duplicated()]\n    final_features_df = add_physics_and_interaction_features(final_features_df)\n    \n    print(f\"Created {final_features_df.shape[1]} features in total.\")\n    return final_features_df.fillna(0).replace([np.inf, -np.inf], 0)\n\ndef official_competition_score(y_true, y_pred, sigma_pred, naive_mean, naive_sigma, fsg_sigma_true=1e-6, airs_sigma_true=1e-5):\n    y_true, y_pred, sigma_pred = np.asarray(y_true), np.asarray(y_pred), np.asarray(sigma_pred)\n    sigma_pred = np.clip(sigma_pred, 1e-15, None)\n    n_wavelengths = y_true.shape[1]\n    \n    if n_wavelengths == 283:\n        sigma_true_arr = np.append(np.array([fsg_sigma_true]), np.ones(n_wavelengths - 1) * airs_sigma_true)\n        FGS1_RELATIVE_WEIGHT = 57.846\n        weights_arr = np.append(np.array([FGS1_RELATIVE_WEIGHT]), np.ones(n_wavelengths - 1))\n    elif n_wavelengths == 1:\n        sigma_true_arr = np.array([fsg_sigma_true])\n        weights_arr = np.ones(n_wavelengths)\n    else:\n        sigma_true_arr = np.ones(n_wavelengths) * airs_sigma_true\n        weights_arr = np.ones(n_wavelengths)\n\n    sigma_true = np.tile(sigma_true_arr, (y_true.shape[0], 1))\n    weights = np.tile(weights_arr, (y_true.shape[0], 1))\n    \n    gll_pred = scipy.stats.norm.logpdf(y_true, loc=y_pred, scale=sigma_pred)\n    gll_true = scipy.stats.norm.logpdf(y_true, loc=y_true, scale=sigma_true)\n    gll_mean = scipy.stats.norm.logpdf(y_true, loc=naive_mean, scale=naive_sigma)\n    ind_scores = (gll_pred - gll_mean) / (gll_true - gll_mean + 1e-9)\n    final_score = np.average(ind_scores, weights=weights)\n    return float(np.clip(final_score, 0.0, 1.0))\n\ndef find_best_calibration(y_true, y_pred, sigma_raw, naive_mean, naive_sigma, instrument_name):\n    print(f\"\\n--- [Validation] Searching for best calibration factors for {instrument_name}... ---\")\n    best_score, best_scaling, best_additive = -1.0, 1.0, 0.0\n    sigma_true_const = 1e-6 if \"FGS\" in instrument_name else 1e-5\n    \n    for scaling in config.CALIBRATION_SCALING_FACTORS:\n        for additive in config.CALIBRATION_ADDITIVE_FACTORS:\n            sigma_calibrated = (sigma_raw * scaling) + additive\n            score = official_competition_score(y_true, y_pred, sigma_calibrated, naive_mean, naive_sigma,\n                                               fsg_sigma_true=sigma_true_const, airs_sigma_true=sigma_true_const)\n            if score > best_score:\n                best_score, best_scaling, best_additive = score, scaling, additive\n    \n    print(f\"--- Best factors for {instrument_name}: Scale={best_scaling}, Add={best_additive} (Best Score: {best_score:.4f}) ---\")\n    return {'scaling': best_scaling, 'additive': best_additive}\n# --------------------------------------------------------------------------------------------------------------\n\nif __name__ == '__main__':\n    ## ======================================================================\n    ## --- SUBMISSION MODE ---\n    ## ======================================================================\n    if not TRAIN:\n        print(\"\\n\" + \"=\"*50 + \"\\n======         SUBMISSION MODE          ======\\n\" + \"=\"*50 + \"\\n\")\n        \n        sample_submission = pd.read_csv(config.SAMPLE_SUBMISSION_PATH, index_col='planet_id')\n        test_star_info_df = pd.read_csv(config.TEST_STAR_INFO_PATH, index_col='planet_id')\n        \n        fgs_signals_test = load_all_observations('test', test_star_info_df.index, 'FGS1')\n        airs_signals_test = load_all_observations('test', test_star_info_df.index, 'AIRS-CH0')\n        test_features_df = combined_feature_engineering(fgs_signals_test, airs_signals_test, test_star_info_df)\n        \n        print(\"Loading saved models and parameters...\")\n        with open(os.path.join(config.OUTPUT_PATH, 'feature_columns.pkl'), 'rb') as f: train_cols = pickle.load(f)\n        with open(os.path.join(config.OUTPUT_PATH, 'calibration_params.pkl'), 'rb') as f: calibration_params = pickle.load(f)\n\n        print(\"Aligning test features with training features...\")\n        test_features_df = test_features_df.reindex(columns=train_cols).fillna(0)\n\n        # --- Predict using the ensemble of models ---\n        all_mu_preds = []\n        all_sigma_preds = []\n        for fold in range(config.N_FOLDS):\n            print(f\"--- Predicting with models from Fold {fold+1}/{config.N_FOLDS} ---\")\n            with open(os.path.join(config.OUTPUT_PATH, f'model_mean_fold_{fold}.pkl'), 'rb') as f: model_mean = pickle.load(f)\n            with open(os.path.join(config.OUTPUT_PATH, f'model_sigma_fold_{fold}.pkl'), 'rb') as f: model_sigma = pickle.load(f)\n            \n            mu_pred = model_mean.predict(test_features_df)\n            sigma_pred = model_sigma.predict(test_features_df)\n            \n            all_mu_preds.append(mu_pred)\n            all_sigma_preds.append(sigma_pred)\n            \n        # --- Average the predictions ---\n        print(\"Averaging predictions from all folds...\")\n        y_pred_test = np.mean(all_mu_preds, axis=0).clip(0, None)\n        sigma_raw_test = np.mean(all_sigma_preds, axis=0).clip(1e-10, None)\n\n        # --- Apply the single, globally optimized calibration ---\n        sigma_pred_test_fgs1 = (sigma_raw_test[:, :1] * calibration_params['fgs1']['scaling']) + calibration_params['fgs1']['additive']\n        sigma_pred_test_airs = (sigma_raw_test[:, 1:] * calibration_params['airs']['scaling']) + calibration_params['airs']['additive']\n        sigma_pred_test = np.hstack([sigma_pred_test_fgs1, sigma_pred_test_airs])\n\n        print(\"Creating submission file...\")\n        pred_df = pd.DataFrame(y_pred_test, index=sample_submission.index, columns=sample_submission.columns[:283])\n        sigma_df = pd.DataFrame(sigma_pred_test, index=sample_submission.index, columns=sample_submission.columns[283:])\n        submission_df = pd.concat([pred_df, sigma_df], axis=1)\n        \n        submission_df.to_csv('submission_xgb.csv')\n        print(submission_df.head())\n        print(\"\\n'submission.csv' created successfully!\")\n\n    ## ======================================================================\n    ## --- TRAINING MODE ---\n    ## ======================================================================\n    else:\n        print(\"\\n\" + \"=\"*50 + \"\\n======          TRAINING MODE           ======\\n\" + \"=\"*50 + \"\\n\")\n        \n        os.makedirs(config.OUTPUT_PATH, exist_ok=True)\n        \n        train_labels_df = pd.read_csv(config.TRAIN_LABELS_PATH, index_col='planet_id')\n        train_star_info_df = pd.read_csv(config.TRAIN_STAR_INFO_PATH, index_col='planet_id').loc[train_labels_df.index]\n        \n        if DEBUG:\n            train_labels_df = train_labels_df.head(100)\n            train_star_info_df = train_star_info_df.head(100)\n\n        fgs_signals_train = load_all_observations('train', train_labels_df.index, 'FGS1')\n        airs_signals_train = load_all_observations('train', train_labels_df.index, 'AIRS-CH0')\n        \n        train_features_df = combined_feature_engineering(fgs_signals_train, airs_signals_train, train_star_info_df)\n        train_labels = train_labels_df.reindex(train_features_df.index).values\n        naive_mu_train, naive_sigma_train = np.mean(train_labels), np.std(train_labels)\n\n        # --- Save feature columns before training ---\n        with open(os.path.join(config.OUTPUT_PATH, 'feature_columns.pkl'), 'wb') as f:\n            pickle.dump(train_features_df.columns.tolist(), f)\n            \n        # --- Prepare for K-Fold training ---\n        # Stratify on a key feature. Binning continuous values is good practice for stratification.\n        stratify_col = pd.cut(train_features_df['fgs_depth_mean'], bins=10, labels=False)\n        skf = StratifiedKFold(n_splits=config.N_FOLDS, shuffle=True, random_state=config.RANDOM_STATE)\n        \n        oof_mu = np.zeros_like(train_labels)\n        oof_sigma = np.zeros_like(train_labels)\n        \n        for fold, (train_idx, val_idx) in enumerate(skf.split(train_features_df, stratify_col)):\n            print(\"\\n\" + \"*\"*20 + f\" FOLD {fold+1}/{config.N_FOLDS} \" + \"*\"*20)\n            \n            # --- Correctly split data using indices ---\n            X_train_f, X_val_f = train_features_df.iloc[train_idx], train_features_df.iloc[val_idx]\n            y_train_f, y_val_f = train_labels[train_idx], train_labels[val_idx]\n\n            # --- Stage 1: Train Mean Model for this fold ---\n            print(f\"--- [Fold {fold+1}] Training Mean (mu) Model... ---\")\n            model_mean = MultiOutputRegressor(xgb.XGBRegressor(**config.XGB_PARAMS_MEAN), n_jobs=-1)\n            model_mean.fit(X_train_f, y_train_f)\n\n            # --- Stage 2: Train Sigma Model for this fold ---\n            print(f\"--- [Fold {fold+1}] Training Sigma (uncertainty) Model... ---\")\n            y_pred_mu_val = model_mean.predict(X_val_f)\n            y_target_sigma_val = np.abs(y_val_f - y_pred_mu_val)\n            \n            model_sigma = MultiOutputRegressor(xgb.XGBRegressor(**config.XGB_PARAMS_SIGMA), n_jobs=-1)\n            model_sigma.fit(X_val_f, y_target_sigma_val)\n\n            # --- Store Out-of-Fold (OOF) predictions for final validation ---\n            oof_mu[val_idx] = y_pred_mu_val\n            oof_sigma[val_idx] = model_sigma.predict(X_val_f)\n            \n            # --- Stage 3: Save the models for this fold ---\n            print(f\"--- [Fold {fold+1}] Saving models... ---\")\n            with open(os.path.join(config.OUTPUT_PATH, f'model_mean_fold_{fold}.pkl'), 'wb') as f:\n                pickle.dump(model_mean, f)\n            with open(os.path.join(config.OUTPUT_PATH, f'model_sigma_fold_{fold}.pkl'), 'wb') as f:\n                pickle.dump(model_sigma, f)\n        \n        # --- Stage 4: Validate and Calibrate using ALL OOF predictions ---\n        print(\"\\n\" + \"=\"*50 + \"\\n======   FINAL OOF VALIDATION & CALIBRATION   ======\\n\" + \"=\"*50 + \"\\n\")\n        \n        # Clip OOF predictions\n        oof_mu_clipped = oof_mu.clip(0, None)\n        oof_sigma_clipped = oof_sigma.clip(1e-10, None)\n\n        # Split OOF predictions by instrument for separate calibration\n        y_true_fgs1, y_true_airs = train_labels[:, :1], train_labels[:, 1:]\n        oof_mu_fgs1, oof_mu_airs = oof_mu_clipped[:, :1], oof_mu_clipped[:, 1:]\n        oof_sigma_fgs1, oof_sigma_airs = oof_sigma_clipped[:, :1], oof_sigma_clipped[:, 1:]\n        \n        best_params_fgs1 = find_best_calibration(y_true_fgs1, oof_mu_fgs1, oof_sigma_fgs1, naive_mu_train, naive_sigma_train, \"FGS1\")\n        best_params_airs = find_best_calibration(y_true_airs, oof_mu_airs, oof_sigma_airs, naive_mu_train, naive_sigma_train, \"AIRS\")\n        \n        # Apply best calibration to OOF sigma to calculate final score\n        oof_sigma_calibrated_fgs1 = (oof_sigma_fgs1 * best_params_fgs1['scaling']) + best_params_fgs1['additive']\n        oof_sigma_calibrated_airs = (oof_sigma_airs * best_params_airs['scaling']) + best_params_airs['additive']\n        oof_sigma_calibrated_total = np.hstack([oof_sigma_calibrated_fgs1, oof_sigma_calibrated_airs])\n        \n        final_cv_score = official_competition_score(train_labels, oof_mu_clipped, oof_sigma_calibrated_total, naive_mu_train, naive_sigma_train)\n        print(f\"\\n--- Final Combined OOF CV Score: {final_cv_score:.5f} ---\")\n        \n        # --- Stage 5: Save the final calibration parameters ---\n        calibration_params = {'fgs1': best_params_fgs1, 'airs': best_params_airs}\n        with open(os.path.join(config.OUTPUT_PATH, 'calibration_params.pkl'), 'wb') as f:\n            pickle.dump(calibration_params, f)\n            \n        print(\"\\nTraining complete. All models and artifacts saved to:\", config.OUTPUT_PATH)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T05:29:19.046977Z","iopub.execute_input":"2025-09-24T05:29:19.0472Z","iopub.status.idle":"2025-09-24T05:31:07.880702Z","shell.execute_reply.started":"2025-09-24T05:29:19.047184Z","shell.execute_reply":"2025-09-24T05:31:07.879972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb = pd.read_csv(\"submission_xgb.csv\")\nnn = pd.read_csv(\"submission_nn.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T05:31:07.882099Z","iopub.execute_input":"2025-09-24T05:31:07.88233Z","iopub.status.idle":"2025-09-24T05:31:07.908391Z","shell.execute_reply.started":"2025-09-24T05:31:07.882311Z","shell.execute_reply":"2025-09-24T05:31:07.907889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Identify columns\nwl_cols    = [c for c in xgb.columns if c.startswith(\"wl_\")]\nsigma_cols = [c for c in xgb.columns if c.startswith(\"sigma_\")]\n\n# Weighted blends\nwl_blend    = 0.2 * xgb[wl_cols] + 0.8 * nn[wl_cols]\nsigma_blend = 0.4 * xgb[sigma_cols] + 0.6 * nn[sigma_cols]\n\n# Combine with planet_id\nsubmission = pd.concat(\n    [xgb[[\"planet_id\"]], wl_blend, sigma_blend],\n    axis=1\n)\n\n# Save final submission\nsubmission.to_csv(\"submission_396.csv\", index=False)\nprint(\"Saved submission.csv with shape:\", submission.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T05:31:07.909056Z","iopub.execute_input":"2025-09-24T05:31:07.909237Z","iopub.status.idle":"2025-09-24T05:31:07.9235Z","shell.execute_reply.started":"2025-09-24T05:31:07.909222Z","shell.execute_reply":"2025-09-24T05:31:07.922831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nh = pd.read_csv(\"submission_396.csv\")\nl = pd.read_csv(\"submission_380.csv\")\n\nwl_cols    = [c for c in h.columns if c.startswith(\"wl_\")]\nsigma_cols = [c for c in h.columns if c.startswith(\"sigma_\")]\n\n# choose a heavier weight for the better file\nw_h = 0.75     # weight for high-score file\nw_l = 0.25     # weight for lower-score file\n\nblend = h.copy()\nblend[wl_cols]    = w_h * h[wl_cols]    + w_l * l[wl_cols]\nblend[sigma_cols] = w_h * h[sigma_cols] + w_l * l[sigma_cols]\n\nblend.to_csv(\"submission.csv\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T05:31:23.949029Z","iopub.execute_input":"2025-09-24T05:31:23.949281Z","iopub.status.idle":"2025-09-24T05:31:24.045515Z","shell.execute_reply.started":"2025-09-24T05:31:23.949263Z","shell.execute_reply":"2025-09-24T05:31:24.044902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"blend","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T05:31:26.279085Z","iopub.execute_input":"2025-09-24T05:31:26.279585Z","iopub.status.idle":"2025-09-24T05:31:26.297376Z","shell.execute_reply.started":"2025-09-24T05:31:26.27956Z","shell.execute_reply":"2025-09-24T05:31:26.296688Z"}},"outputs":[],"execution_count":null}]}