{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.13"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":101849,"databundleVersionId":13093295,"sourceType":"competition"},{"sourceId":9629432,"sourceType":"datasetVersion","datasetId":5846888},{"sourceId":562901,"sourceType":"modelInstanceVersion","modelInstanceId":425925,"modelId":443413},{"sourceId":562906,"sourceType":"modelInstanceVersion","modelInstanceId":425929,"modelId":443416}],"dockerImageVersionId":31090,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":25.428306,"end_time":"2025-09-19T07:12:03.82627","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-09-19T07:11:38.397964","version":"2.6.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"0221a12bf8b641a88c27a35a22782791":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"08dfcda2da794462b3ee882ccd02f395":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"0b8a058678604ebbbbbdc21cb6610a88":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_941a807a0d194965b7c51bceb9f95ac9","IPY_MODEL_236d4b4543f247c8aba6614c0bcfd867","IPY_MODEL_58555c7a27e247d482e21996c528c9fc"],"layout":"IPY_MODEL_0f49c451399942a08b0bc4ac57e9cf24","tabbable":null,"tooltip":null}},"0b9906a7b1ba4567babb37c2d60aa1f4":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"0cf46e951a77416881d5290737a4fa5a":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"0f49c451399942a08b0bc4ac57e9cf24":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"1421eb6bef4e431ca6e0260b61658934":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"14ae4578043d4771bef60ed3cbea931c":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_4c7bf3feae784d8eba4158ac99e586f0","placeholder":"​","style":"IPY_MODEL_2350cec3b6e94e5f9d134ceb728e3836","tabbable":null,"tooltip":null,"value":" 1/1 [00:01&lt;00:00,  1.95s/it]"}},"189fe9759ac44818b6f02937e2b286f2":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"1d1d7afb980b48d9994b27a732b5b369":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_90b006b9e7944f22a9d69dc984ede712","IPY_MODEL_e23a77085f114dafa9f74cbe01a763eb","IPY_MODEL_7296539cae04499aa1ab1d03eacb7dab"],"layout":"IPY_MODEL_d24efe355daa44d6b5178bcdd723fd0e","tabbable":null,"tooltip":null}},"212db83369c44a829d1d576ffdfdfa4b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_aefa454f0ae444898a4951ef5f85934f","IPY_MODEL_6c24188c13c44ffa8b393adaf5985f09","IPY_MODEL_590b6380947c4a9e91011350dad78521"],"layout":"IPY_MODEL_de6bcf4060574b48a821d3f4b7ddf226","tabbable":null,"tooltip":null}},"21cbb4523c5642edba4a5fe2ab711e77":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_e40ba11019ee4bdb87d1923a283bf0ba","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_08dfcda2da794462b3ee882ccd02f395","tabbable":null,"tooltip":null,"value":1}},"2350cec3b6e94e5f9d134ceb728e3836":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"236d4b4543f247c8aba6614c0bcfd867":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_c16112696bf94043b9c5cf33c94b83ad","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_454e4aabdbbf47ec9d430ef1ef114b88","tabbable":null,"tooltip":null,"value":1}},"24dc4562f6394bafb93c6ed818638916":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_44f2827e1cfc405eb82b62ef5dbc935f","placeholder":"​","style":"IPY_MODEL_8658c5da8cb04977aff947507178bb38","tabbable":null,"tooltip":null,"value":"QUEUEING TASKS | : 100%"}},"269462dbae06474e9ee44a3f9d228c61":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"2a3a314647af4bf78305d68c5bb54276":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_ba1c300814c244319a89907365905c40","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_9423a83484c643e2b610c564072711ac","tabbable":null,"tooltip":null,"value":1}},"2b82dc160b4948aab93a0a80105f3ed2":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"2d7f5dc948314103ab32dfe56546fd03":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_d4bd3f4ad9e34e1fb18cac2619ee6486","placeholder":"​","style":"IPY_MODEL_68b4ab438c2c420d8e7c1eaac7afbd68","tabbable":null,"tooltip":null,"value":"PROCESSING TASKS | : 100%"}},"30a1464f20a642f0b11b7091a9f37bb8":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"354005cbb8fb4c84a75112772a55306d":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_24dc4562f6394bafb93c6ed818638916","IPY_MODEL_21cbb4523c5642edba4a5fe2ab711e77","IPY_MODEL_90ef9b0099b04f7ebbc1e8fc221d1bd6"],"layout":"IPY_MODEL_849be3ebb49043d0a0c9cc287dc5a3d0","tabbable":null,"tooltip":null}},"44f2827e1cfc405eb82b62ef5dbc935f":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"454e4aabdbbf47ec9d430ef1ef114b88":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"4aecfd1e48a94a67a7666423c31193e7":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_2d7f5dc948314103ab32dfe56546fd03","IPY_MODEL_2a3a314647af4bf78305d68c5bb54276","IPY_MODEL_14ae4578043d4771bef60ed3cbea931c"],"layout":"IPY_MODEL_c89e08166c024dd3a0dd0427ba229082","tabbable":null,"tooltip":null}},"4c7bf3feae784d8eba4158ac99e586f0":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"512addfc7bde4e1d9060aeb8cf7121f5":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"58555c7a27e247d482e21996c528c9fc":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_0221a12bf8b641a88c27a35a22782791","placeholder":"​","style":"IPY_MODEL_30a1464f20a642f0b11b7091a9f37bb8","tabbable":null,"tooltip":null,"value":" 1/1 [00:03&lt;00:00,  3.29s/it]"}},"590b6380947c4a9e91011350dad78521":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_512addfc7bde4e1d9060aeb8cf7121f5","placeholder":"​","style":"IPY_MODEL_5b97b87685574e30aeb301c8daa25bf0","tabbable":null,"tooltip":null,"value":" 1/1 [00:00&lt;00:00, 116.42it/s]"}},"5a8d8fede7b147d99b7d992389f93d39":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"5ade02a1c1114e0dbc9e2aa6b5b9a94a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"5b97b87685574e30aeb301c8daa25bf0":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"68b4ab438c2c420d8e7c1eaac7afbd68":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"6c24188c13c44ffa8b393adaf5985f09":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_deeca099ce7d4b7abdf7d38cd4cc857f","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_5ade02a1c1114e0dbc9e2aa6b5b9a94a","tabbable":null,"tooltip":null,"value":1}},"6c4c8fe86e314382b00d72c34d49c3e3":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"7296539cae04499aa1ab1d03eacb7dab":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_b48ede0d236a4bfaa93d32e59fdfa3cc","placeholder":"​","style":"IPY_MODEL_b0e6f779c11d49d180ef7898ac8db463","tabbable":null,"tooltip":null,"value":" 1/1 [00:00&lt;00:00, 120.98it/s]"}},"7e95ba94c897445ab8e4f423dd21edf3":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"849be3ebb49043d0a0c9cc287dc5a3d0":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8658c5da8cb04977aff947507178bb38":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"8ca91ba236334f629b2e5bb5e3a00017":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8f6c667a843541cda0989163f93d692f":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"90b006b9e7944f22a9d69dc984ede712":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_2b82dc160b4948aab93a0a80105f3ed2","placeholder":"​","style":"IPY_MODEL_189fe9759ac44818b6f02937e2b286f2","tabbable":null,"tooltip":null,"value":"COLLECTING RESULTS | : 100%"}},"90ef9b0099b04f7ebbc1e8fc221d1bd6":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_1421eb6bef4e431ca6e0260b61658934","placeholder":"​","style":"IPY_MODEL_5a8d8fede7b147d99b7d992389f93d39","tabbable":null,"tooltip":null,"value":" 1/1 [00:00&lt;00:00, 106.84it/s]"}},"93157a829c464d17820735b94215f356":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_a8d91662839743cebcbe9093c539bb3b","placeholder":"​","style":"IPY_MODEL_cf92069fd2b84c71ba91ec2293079823","tabbable":null,"tooltip":null,"value":" 1/1 [00:00&lt;00:00, 124.80it/s]"}},"941a807a0d194965b7c51bceb9f95ac9":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_8f6c667a843541cda0989163f93d692f","placeholder":"​","style":"IPY_MODEL_ba2b21785a1b4e4ebd7743ecb6d5ef17","tabbable":null,"tooltip":null,"value":"PROCESSING TASKS | : 100%"}},"9423a83484c643e2b610c564072711ac":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"9d4b0186bd6e46a18ccd32d24d49e285":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_cd2488be75eb409a9efb2caedcedda7d","IPY_MODEL_af747d08a4db47e8969fd622d3ba9bc1","IPY_MODEL_93157a829c464d17820735b94215f356"],"layout":"IPY_MODEL_9e1848968a344197a2f802904c9b65c9","tabbable":null,"tooltip":null}},"9e1848968a344197a2f802904c9b65c9":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"a8d91662839743cebcbe9093c539bb3b":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"aefa454f0ae444898a4951ef5f85934f":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_8ca91ba236334f629b2e5bb5e3a00017","placeholder":"​","style":"IPY_MODEL_c52fbf3fd8cd4288b112c04ea561e608","tabbable":null,"tooltip":null,"value":"QUEUEING TASKS | : 100%"}},"af747d08a4db47e8969fd622d3ba9bc1":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_269462dbae06474e9ee44a3f9d228c61","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_0b9906a7b1ba4567babb37c2d60aa1f4","tabbable":null,"tooltip":null,"value":1}},"b0e6f779c11d49d180ef7898ac8db463":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"b48ede0d236a4bfaa93d32e59fdfa3cc":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"ba1c300814c244319a89907365905c40":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"ba2b21785a1b4e4ebd7743ecb6d5ef17":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"c16112696bf94043b9c5cf33c94b83ad":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"c52fbf3fd8cd4288b112c04ea561e608":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"c89e08166c024dd3a0dd0427ba229082":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"cd2488be75eb409a9efb2caedcedda7d":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_6c4c8fe86e314382b00d72c34d49c3e3","placeholder":"​","style":"IPY_MODEL_d548b62647564e2988b3bbfe4d496a71","tabbable":null,"tooltip":null,"value":"COLLECTING RESULTS | : 100%"}},"cf92069fd2b84c71ba91ec2293079823":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"d24efe355daa44d6b5178bcdd723fd0e":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"d4bd3f4ad9e34e1fb18cac2619ee6486":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"d548b62647564e2988b3bbfe4d496a71":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"de6bcf4060574b48a821d3f4b7ddf226":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"deeca099ce7d4b7abdf7d38cd4cc857f":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"e23a77085f114dafa9f74cbe01a763eb":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_0cf46e951a77416881d5290737a4fa5a","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_7e95ba94c897445ab8e4f423dd21edf3","tabbable":null,"tooltip":null,"value":1}},"e40ba11019ee4bdb87d1923a283bf0ba":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install --no-index --find-links=/kaggle/input/ariel-2024-pqdm pqdm > /dev/null","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":4.302424,"end_time":"2025-09-19T07:11:47.255904","exception":false,"start_time":"2025-09-19T07:11:42.95348","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T16:02:39.711948Z","iopub.execute_input":"2025-09-24T16:02:39.71224Z","iopub.status.idle":"2025-09-24T16:02:44.722928Z","shell.execute_reply.started":"2025-09-24T16:02:39.712218Z","shell.execute_reply":"2025-09-24T16:02:44.721887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn.functional as F\nimport multiprocessing as mp\nimport torch.nn as nn\nimport os\nimport matplotlib.pyplot as plt\nimport itertools\nfrom tqdm import tqdm\nfrom pqdm.threads import pqdm\nfrom astropy.stats import sigma_clip\nfrom scipy.optimize import minimize\nfrom torch.utils.data import DataLoader, TensorDataset, random_split\nfrom sklearn.preprocessing import StandardScaler\nfrom scipy.signal import savgol_filter\nfrom sklearn.metrics import mean_squared_error","metadata":{"papermill":{"duration":8.611729,"end_time":"2025-09-19T07:11:55.870946","exception":false,"start_time":"2025-09-19T07:11:47.259217","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T16:02:44.724564Z","iopub.execute_input":"2025-09-24T16:02:44.72491Z","iopub.status.idle":"2025-09-24T16:02:51.685074Z","shell.execute_reply.started":"2025-09-24T16:02:44.72488Z","shell.execute_reply":"2025-09-24T16:02:51.684318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================== #\n#  Ariel 2025 — end-to-end script #\n#  Train GLL -> tune σ -> Submit  #\n#  (with tqdm logging + 10% debug)#\n# =============================== #\n\nimport os, gc, math, warnings, itertools\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom itertools import product\n# tqdm\nfrom tqdm.auto import tqdm, trange\n\n# SciPy / Astropy\nfrom scipy.signal import savgol_filter\nfrom scipy.optimize import minimize\ntry:\n    from astropy.stats import sigma_clip\nexcept Exception:\n    sigma_clip = None  # fallback\n\n# Torch\nimport torch\nimport torch.nn as nn\n\n# -------------------- Debug Switch -------------------- #\nDEBUG_MODE = False    # ← False로 두면 전체 수행\nDEBUG_FRAC = 1     # ← 10%만 샘플링\nDEBUG_SEED = 42\n\nGRID_F_DEBUG = np.linspace(0.8, 1.4, 9)\nGRID_A_DEBUG = np.linspace(0.8, 1.4, 9)\n\n# -------------------- Config -------------------- #\nROOT_PATH = \"/kaggle/input/ariel-data-challenge-2025\"\n\nclass Config:\n    DATA_PATH = ROOT_PATH\n    SCALE = 0.946\n    SIGMA = 0.00056\n\n    CUT_INF = 39\n    CUT_SUP = 321\n\n    SENSOR_CONFIG = {\n        \"AIRS-CH0\": {\n            \"raw_shape\": [11250, 32, 356],\n            \"calibrated_shape\": [1, 32, CUT_SUP - CUT_INF],\n            \"linear_corr_shape\": (6, 32, 356),\n            \"dt_pattern\": (0.1, 4.5),\n            \"binning\": 30\n        },\n        \"FGS1\": {\n            \"raw_shape\": [135000, 32, 32],\n            \"calibrated_shape\": [1, 32, 32],\n            \"linear_corr_shape\": (6, 32, 32),\n            \"dt_pattern\": (0.1, 0.1),\n            \"binning\": 30 * 12\n        }\n    }\n\n    MODEL_PHASE_DETECTION_SLICE = slice(30, 140)\n    MODEL_OPTIMIZATION_DELTA = 11\n    MODEL_POLYNOMIAL_DEGREE = 3\n\n    N_JOBS = -1\n\n# -------------------- Small logger -------------------- #\ndef log(msg: str):\n    \"\"\"Safe logging that doesn't break tqdm bars.\"\"\"\n    tqdm.write(str(msg))\n\n# -------------------- Utils -------------------- #\ndef _phase_detector_signal(signal, cfg: Config):\n    sl = cfg.MODEL_PHASE_DETECTION_SLICE\n    min_idx = int(np.argmin(signal[sl])) + sl.start\n    s1 = signal[:min_idx]; s2 = signal[min_idx:]\n    if s1.size < 3 or s2.size < 3:\n        return 0, len(signal) - 1\n    g1 = np.gradient(s1); g1_max = np.max(g1) if g1.size else 0.0\n    g2 = np.gradient(s2); g2_max = np.max(g2) if g2.size else 0.0\n    if g1_max != 0: g1 /= g1_max\n    if g2_max != 0: g2 /= g2_max\n    phase1 = int(np.argmin(g1))\n    phase2 = int(np.argmax(g2)) + min_idx\n    return phase1, phase2\n\ndef estimate_sigma_fgs(preprocessed_data, cfg: Config):\n    sig_rel = []\n    delta = cfg.MODEL_OPTIMIZATION_DELTA\n    eps = 1e-12\n    for single in tqdm(preprocessed_data, desc=\"σ-estimate FGS (per-planet)\", dynamic_ncols=True, leave=False):\n        air_white = savgol_filter(single[:, 1:].mean(axis=1), 20, 2)\n        p1, p2 = _phase_detector_signal(air_white, cfg)\n        p1 = max(delta, p1); p2 = min(len(air_white) - delta - 1, p2)\n        fgs = single[:, 0]\n        oot = (fgs[: p1 - delta] if p1 - delta > 0 else np.empty(0, fgs.dtype))\n        if p2 + delta < fgs.size:\n            oot = np.concatenate([oot, fgs[p2 + delta :]])\n        inn = fgs[p1 + delta : max(p1 + delta, p2 - delta)]\n        if oot.size == 0 or inn.size == 0:\n            sig_rel.append(np.nan); continue\n        n_oot, n_in = len(oot), len(inn)\n        var_oot = np.nanvar(oot, ddof=1); var_in = np.nanvar(inn, ddof=1)\n        oot_mean = float(np.nanmean(oot)) if np.isfinite(np.nanmean(oot)) else float(np.nanmean(fgs))\n        sigma_rel = np.sqrt(var_oot/max(n_oot,1) + var_in/max(n_in,1)) / max(oot_mean, eps)\n        sig_rel.append(sigma_rel)\n    s = np.asarray(sig_rel, dtype=float)\n    mask = np.isfinite(s) & (s > 0)\n    med = float(np.nanmedian(s[mask])) if mask.any() else 1.0\n    k = np.ones_like(s)\n    if med > 0 and np.isfinite(med):\n        k[mask] = np.sqrt(s[mask] / med)\n    k = np.clip(k, 0.85, 1.30)\n    return k * cfg.SIGMA * 1.04\n\ndef estimate_sigma_air(preprocessed_data, cfg: Config):\n    sig_rel = []\n    delta = cfg.MODEL_OPTIMIZATION_DELTA\n    eps = 1e-12\n    for single in tqdm(preprocessed_data, desc=\"σ-estimate AIRS (per-planet)\", dynamic_ncols=True, leave=False):\n        white = np.nanmean(single[:, 1:], axis=1)\n        white_s = savgol_filter(white, 20, 2)\n        p1, p2 = _phase_detector_signal(white_s, cfg)\n        p1 = max(delta, p1); p2 = min(len(white) - delta - 1, p2)\n        oot_left = white[: p1 - delta] if p1 - delta > 0 else np.empty(0, white.dtype)\n        oot_right = white[p2 + delta :] if (p2 + delta) < white.size else np.empty(0, white.dtype)\n        oot = np.concatenate([oot_left, oot_right]) if (oot_left.size + oot_right.size) else oot_left\n        inn = white[p1 + delta : max(p1 + delta, p2 - delta)]\n        if oot.size == 0 or inn.size == 0:\n            sig_rel.append(np.nan); continue\n        n_oot, n_in = len(oot), len(inn)\n        var_oot = np.nanvar(oot, ddof=1); var_in = np.nanvar(inn, ddof=1)\n        oot_mean = float(np.nanmean(oot)) if np.isfinite(np.nanmean(oot)) else float(np.nanmean(white))\n        sigma_rel = np.sqrt(var_oot/max(n_oot,1) + var_in/max(n_in,1)) / max(oot_mean, eps)\n        sig_rel.append(sigma_rel)\n    s = np.asarray(sig_rel, dtype=float)\n    mask = np.isfinite(s) & (s > 0)\n    med = float(np.nanmedian(s[mask])) if mask.any() else 1.0\n    k = np.ones_like(s)\n    if med > 0 and np.isfinite(med):\n        k[mask] = np.sqrt(s[mask] / med)\n    k = np.clip(k, 0.92, 1.22)\n    return k * cfg.SIGMA * 1.04\n\n# -------------------- IO / Processing -------------------- #\ndef _sample_ids(ids, frac=1.0, seed=42):\n    if not (0 < frac < 1.0):\n        return list(ids)\n    rng = np.random.RandomState(seed)\n    n = max(1, int(len(ids)*frac))\n    sel = rng.choice(ids, size=n, replace=False)\n    return sorted(sel.tolist())\n\nclass SignalProcessor:\n    def __init__(self, config: Config, split: str):\n        self.cfg = config\n        self.split = split  # 'train' or 'test'\n        self.adc_info = pd.read_csv(f\"{self.cfg.DATA_PATH}/adc_info.csv\")\n        star = pd.read_csv(f'{self.cfg.DATA_PATH}/{split}_star_info.csv', index_col='planet_id')\n        self.planet_ids = star.index.astype(int).tolist()\n\n    def _apply_linear_corr(self, linear_corr, signal):\n        coeffs = np.flip(linear_corr, axis=0)\n        x = signal.astype(np.float64, copy=False)\n        out = np.empty_like(x, dtype=np.float64)\n        out[...] = coeffs[0]\n        for k in range(1, coeffs.shape[0]):\n            np.multiply(out, x, out=out)\n            out += coeffs[k]\n        return out.astype(signal.dtype, copy=False)\n\n    def _calibrate_single_signal(self, planet_id, sensor):\n        cfg = self.cfg.SENSOR_CONFIG[sensor]\n        base = f\"{self.cfg.DATA_PATH}/{self.split}/{planet_id}\"\n\n        signal = pd.read_parquet(f\"{base}/{sensor}_signal_0.parquet\").to_numpy()\n        dark   = pd.read_parquet(f\"{base}/{sensor}_calibration_0/dark.parquet\").to_numpy()\n        dead   = pd.read_parquet(f\"{base}/{sensor}_calibration_0/dead.parquet\").to_numpy()\n        flat   = pd.read_parquet(f\"{base}/{sensor}_calibration_0/flat.parquet\").to_numpy()\n        lc     = pd.read_parquet(f\"{base}/{sensor}_calibration_0/linear_corr.parquet\").values.astype(np.float64).reshape(cfg[\"linear_corr_shape\"])\n\n        signal = signal.reshape(cfg[\"raw_shape\"])\n        gain = self.adc_info[f\"{sensor}_adc_gain\"].iloc[0]\n        offset = self.adc_info[f\"{sensor}_adc_offset\"].iloc[0]\n        signal = signal / gain + offset\n\n        if sigma_clip is not None:\n            hot = sigma_clip(dark, sigma=5, maxiters=5).mask\n        else:\n            hot = np.zeros_like(dark, dtype=bool)\n\n        if sensor == \"AIRS-CH0\":\n            signal = signal[:, :, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            lc     = lc[:, :, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            dark   = dark[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            dead   = dead[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            flat   = flat[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            hot    = hot[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n\n        if sensor == \"FGS1\":\n            y0, y1, x0, x1 = 10, 22, 10, 22\n            signal = signal[:, y0:y1, x0:x1]\n            dark   = dark[y0:y1, x0:x1]\n            dead   = dead[y0:y1, x0:x1]\n            flat   = flat[y0:y1, x0:x1]\n            lc     = lc[:,   y0:y1, x0:x1]\n            hot    = hot[y0:y1, x0:x1]\n\n        np.maximum(signal, 0, out=signal)\n\n        if sensor == \"FGS1\":\n            signal = self._apply_linear_corr(lc, signal)\n        elif sensor == \"AIRS-CH0\":\n            sl = (slice(None), slice(10, 22), slice(None))\n            signal[sl] = self._apply_linear_corr(lc[:, 10:22, :], signal[sl])\n        else:\n            signal = self._apply_linear_corr(lc, signal)\n\n        base_dt, inc = cfg[\"dt_pattern\"]\n        signal[::2] -= dark * base_dt\n        signal[1::2] -= dark * (base_dt + inc)\n        return signal\n\n    def _preprocess_calibrated_signal(self, calibrated_signal, sensor):\n        cfg = self.cfg.SENSOR_CONFIG[sensor]\n        binning = cfg[\"binning\"]\n\n        if sensor == \"AIRS-CH0\":\n            roi = calibrated_signal[:, 10:22, :]\n        else:  # FGS1\n            roi = calibrated_signal[:, 10:22, 10:22].reshape(calibrated_signal.shape[0], -1)\n\n        mean_signal = np.nanmean(roi, axis=1)\n        cds_signal = mean_signal[1::2] - mean_signal[0::2]\n\n        n_bins = cds_signal.shape[0] // binning\n        binned = np.array([cds_signal[j*binning:(j+1)*binning].mean(axis=0) for j in range(n_bins)])\n\n        if sensor == \"AIRS-CH0\":\n            q_lo = np.nanpercentile(binned, 5.0, axis=1, keepdims=True)\n            q_hi = np.nanpercentile(binned, 95.0, axis=1, keepdims=True)\n            np.clip(binned, q_lo, q_hi, out=binned)\n            var = np.nanvar(binned, axis=0, ddof=1)\n            med = np.nanmedian(var)\n            safe_var = np.where(~np.isfinite(var) | (var <= 0), med if (np.isfinite(med) and med > 0) else 1.0, var)\n            w = 1.0 / safe_var\n            lo, hi = np.nanpercentile(w, 5.0), np.nanpercentile(w, 95.0)\n            if np.isfinite(lo) and np.isfinite(hi) and lo < hi:\n                w = np.clip(w, lo, hi)\n            s = np.nansum(w)\n            w = w * (binned.shape[1] / s) if np.isfinite(s) and s > 0 else np.ones_like(w)\n            binned *= w[None, :]\n\n        if sensor == \"FGS1\":\n            binned = binned.reshape((binned.shape[0], 1))\n\n        return binned\n\n    def process_all_data(self, frac=1.0, seed=42):\n        used_ids = _sample_ids(self.planet_ids, frac=frac, seed=seed)\n        log(f\"▶ [{self.split}] preprocess {len(used_ids)}/{len(self.planet_ids)} planets (frac={frac})\")\n        pre_fgs, pre_air = [], []\n        for pid in tqdm(used_ids, desc=f\"Calibrate & preprocess {self.split}\", dynamic_ncols=True):\n            cal_fgs = self._calibrate_single_signal(pid, \"FGS1\")\n            pre_fgs.append(self._preprocess_calibrated_signal(cal_fgs, \"FGS1\"))\n            cal_air = self._calibrate_single_signal(pid, \"AIRS-CH0\")\n            pre_air.append(self._preprocess_calibrated_signal(cal_air, \"AIRS-CH0\"))\n        preprocessed = np.concatenate([np.stack(pre_fgs), np.stack(pre_air)], axis=2)\n        log(f\"✔ [{self.split}] preprocessed shape = {preprocessed.shape}  (bins, chans=FGS+282)\")\n        return preprocessed, used_ids\n\n# -------------------- TransitModel -------------------- #\nclass TransitModel:\n    def __init__(self, config: Config):\n        self.cfg = config\n\n    def _phase_detector(self, signal):\n        search_slice = self.cfg.MODEL_PHASE_DETECTION_SLICE\n        min_index = np.argmin(signal[search_slice]) + search_slice.start\n        s1, s2 = signal[:min_index], signal[min_index:]\n        g1 = np.gradient(s1); g2 = np.gradient(s2)\n        g1 = g1 / (g1.max() + 1e-12); g2 = g2 / (g2.max() + 1e-12)\n        p1 = int(np.argmin(g1)); p2 = int(np.argmax(g2)) + min_index\n        return p1, p2\n\n    def _objective_function(self, s, signal, p1, p2):\n        delta = self.cfg.MODEL_OPTIMIZATION_DELTA\n        power = self.cfg.MODEL_POLYNOMIAL_DEGREE\n        if p1 - delta <= 0 or p2 + delta >= len(signal) or p2 - delta - (p1 + delta) < 5:\n            delta = 2\n        y = np.concatenate([\n            signal[: p1 - delta],\n            signal[p1 + delta : p2 - delta] * (1 + s),\n            signal[p2 + delta :]\n        ])\n        x = np.arange(len(y))\n        coeffs = np.polyfit(x, y, deg=power)\n        poly = np.poly1d(coeffs)\n        return float(np.abs(poly(x) - y).mean())\n\n    def predict(self, single_preprocessed_signal):\n        s1d = savgol_filter(single_preprocessed_signal[:, 1:].mean(axis=1), 20, 2)\n        p1, p2 = self._phase_detector(s1d)\n        p1 = max(self.cfg.MODEL_OPTIMIZATION_DELTA, p1)\n        p2 = min(len(s1d) - self.cfg.MODEL_OPTIMIZATION_DELTA - 1, p2)\n        res = minimize(self._objective_function, x0=[0.0001], args=(s1d, p1, p2), method=\"Nelder-Mead\")\n        return float(res.x[0])\n\n    def predict_all(self, preprocessed_signals):\n        preds = []\n        for sp in tqdm(preprocessed_signals, desc=\"TransitModel fit (per-planet)\", dynamic_ncols=True):\n            preds.append(self.predict(sp))\n        arr = np.array(preds) * self.cfg.SCALE\n        log(f\"✔ TransitModel predictions shape = {arr.shape}\")\n        return arr\n\n# -------------------- Simple ResNet-MLPs -------------------- #\nclass ResidualBlock(nn.Module):\n    def __init__(self, dim, p=0.2):\n        super().__init__()\n        self.fc1 = nn.Linear(dim, dim)\n        self.bn1 = nn.BatchNorm1d(dim)\n        self.fc2 = nn.Linear(dim, dim)\n        self.bn2 = nn.BatchNorm1d(dim)\n        self.relu = nn.ReLU()\n        self.dropout = nn.Dropout(p)\n    def forward(self, x):\n        identity = x\n        out = self.relu(self.bn1(self.fc1(x)))\n        out = self.dropout(out)\n        out = self.bn2(self.fc2(out))\n        return self.relu(out + identity)\n\nclass ResNetMLP(nn.Module):\n    def __init__(self, input_dim=3, hidden_dim=32, output_dim=1, num_blocks=3, dropout_rate=0.2):\n        super().__init__()\n        self.input_layer = nn.Linear(input_dim, hidden_dim)\n        self.blocks = nn.Sequential(*[ResidualBlock(hidden_dim, p=dropout_rate) for _ in range(num_blocks)])\n        self.output_layer = nn.Linear(hidden_dim, output_dim)\n    def forward(self, x):\n        x = self.input_layer(x); x = self.blocks(x); x = self.output_layer(x)\n        return x\n\nclass ResNetMLP2(nn.Module):\n    def __init__(self, input_dim=3, hidden_dim=128, output_dim=282, num_blocks=3, dropout_rate=0.2):\n        super().__init__()\n        self.input_layer = nn.Linear(input_dim, hidden_dim)\n        self.blocks = nn.Sequential(*[ResidualBlock(hidden_dim, p=dropout_rate) for _ in range(num_blocks)])\n        self.output_layer = nn.Linear(hidden_dim, output_dim)\n    def forward(self, x):\n        x = self.input_layer(x); x = self.blocks(x); x = self.output_layer(x)\n        return x\n\ndef load_models():\n    log(\"▶ Load pretrained MLPs (FGS, AIRS)\")\n    fgs = ResNetMLP(num_blocks=80, dropout_rate=0.2)\n    fgs.load_state_dict(torch.load(\"/kaggle/input/fgs1/pytorch/default/1/best_model.pth\", map_location=\"cpu\"))\n    fgs.eval()\n    airs = ResNetMLP2(num_blocks=80, dropout_rate=0.3)\n    airs.load_state_dict(torch.load(\"/kaggle/input/airs/pytorch/default/1/best_model_airs.pth\", map_location=\"cpu\"))\n    airs.eval()\n    log(\"✔ Models loaded\")\n    return fgs, airs\n\n# -------------------- Metric (GLL) -------------------- #\nfrom scipy.stats import norm\n\ndef gll_score_numpy(y_true, y_pred, sigma_pred,\n                    naive_mean, naive_sigma,\n                    fgs_sigma_true=1e-6, airs_sigma_true=1e-5, fgs_weight=0.4):\n    sigma_pred = np.clip(sigma_pred, 1e-15, None)\n    n, L = sigma_pred.shape\n    sigma_true = np.append([fgs_sigma_true], np.full(L-1, airs_sigma_true))\n    sigma_true = np.tile(sigma_true, (n, 1))\n    weights = np.append([fgs_weight], np.ones(L-1))\n    weights = np.tile(weights, (n, 1))\n    gll_pred  = norm.logpdf(y_true, loc=y_pred, scale=sigma_pred)\n    gll_true  = norm.logpdf(y_true, loc=y_true, scale=sigma_true)\n    gll_naive = norm.logpdf(y_true, loc=naive_mean, scale=naive_sigma)\n    ind = (gll_pred - gll_naive) / (gll_true - gll_naive + 1e-12)\n    return float(np.clip(np.average(ind, weights=weights), 0.0, 1.0))\n\n# -------------------- Runner -------------------- #\ndef run_pipeline(split: str, cfg: Config, fgs_model, airs_model, frac=1.0, seed=42):\n    \"\"\"전처리 -> TransitModel -> (FGS/AIRS)MLP -> σ 추정 -> (mu, sigma, ids, meta) 반환\"\"\"\n    sp = SignalProcessor(cfg, split)\n    preproc, used_ids = sp.process_all_data(frac=frac, seed=seed)\n\n    tm = TransitModel(cfg)\n    td = tm.predict_all(preproc)   # (N,)\n\n    # Star info\n    log(f\"▶ [{split}] build NN inputs\")\n    star = pd.read_csv(f\"{cfg.DATA_PATH}/{split}_star_info.csv\")\n    star[\"planet_id\"] = star[\"planet_id\"].astype(int)\n    star = star.set_index(\"planet_id\").loc[used_ids]  # reorder\n\n    inp = pd.DataFrame(index=star.index)\n    inp[\"transit_depth\"] = td * 10000.0\n    inp[\"Rs\"] = star[\"Rs\"].values\n    inp[\"i\"]  = star[\"i\"].values\n    X = torch.tensor(inp[[\"transit_depth\", \"Rs\", \"i\"]].values.astype(\"float32\"))\n\n    # NN inference\n    log(f\"▶ [{split}] NN inference (FGS/AIRS)\")\n    with torch.no_grad():\n        pred_fgs  = (fgs_model(X).numpy().reshape(-1) / 10000.0)\n        pred_airs = (airs_model(X).numpy() / 10000.0)  # (N, 282)\n\n    mu = np.concatenate([pred_fgs.reshape(-1,1), pred_airs], axis=1)  # (N, 283)\n    log(f\"✔ [{split}] μ shape = {mu.shape}\")\n\n    # σ\n    log(f\"▶ [{split}] estimate σ\")\n    sigma_fgs_vec = estimate_sigma_fgs(preproc, cfg).reshape(-1)      # (N,)\n    sigma_air_vec = estimate_sigma_air(preproc, cfg).reshape(-1)      # (N,)\n    sigma = np.tile(sigma_air_vec.reshape(-1,1), (1, mu.shape[1]))\n    sigma[:,0] = sigma_fgs_vec\n    log(f\"✔ [{split}] σ shape = {sigma.shape}\")\n\n    return mu.astype(np.float64), sigma.astype(np.float64), used_ids, star, preproc\n\ndef apply_params(mu, sigma, cf=1.0, ca=1.0, mf=1.0, ma=1.0):\n    \"\"\"FGS/AIRS에 각각 다른 μ·σ 스케일을 적용.\"\"\"\n    mu2 = mu.copy()\n    mu2[:, 0]  = np.clip(cf * mu2[:, 0],  0.0, None)     # FGS μ\n    mu2[:, 1:] = np.clip(ca * mu2[:, 1:], 0.0, None)     # AIRS μ\n    s2 = sigma.copy()\n    s2[:, 0]  = np.clip(mf * s2[:, 0],  1e-12, None)     # FGS σ\n    s2[:, 1:] = np.clip(ma * s2[:, 1:], 1e-12, None)     # AIRS σ\n    return mu2, s2\n\ndef _eval_gll_with_params(y_true, mu, sigma, naive_mean, naive_sigma,\n                          cf, ca, mf, ma):\n    mu2, s2 = apply_params(mu, sigma, cf, ca, mf, ma)\n    return gll_score_numpy(y_true, mu2, s2, naive_mean, naive_sigma)\n\ndef tune_mu_sigma_on_train(y_true, mu, sigma,\n                           naive_mean, naive_sigma,\n                           grid_cf=None, grid_ca=None, grid_mf=None, grid_ma=None,\n                           seed=42, max_eval_planets=120, refine=True):\n    \"\"\"\n    4개 파라미터(cf, ca, mf, ma) 그리드 탐색으로 GLL 최대화.\n    - speed를 위해 1차 탐색은 행성 일부(subset)로 수행 후,\n      최적 근방에서 소규모 재탐색(refine) 가능.\n    \"\"\"\n    rng = np.random.RandomState(seed)\n    n = y_true.shape[0]\n    # 1차 탐색은 subset으로 가속\n    if max_eval_planets is not None and n > max_eval_planets:\n        idx = np.sort(rng.choice(n, size=max_eval_planets, replace=False))\n        y_sub, mu_sub, s_sub = y_true[idx], mu[idx], sigma[idx]\n    else:\n        idx = None\n        y_sub, mu_sub, s_sub = y_true, mu, sigma\n\n    # 기본 그리드\n    if grid_cf is None: grid_cf = np.linspace(0.90, 1.10, 9)  # μ FGS\n    if grid_ca is None: grid_ca = np.linspace(0.90, 1.10, 9)  # μ AIRS\n    if grid_mf is None: grid_mf = np.linspace(0.60, 1.80, 25) # σ FGS\n    if grid_ma is None: grid_ma = np.linspace(0.60, 1.80, 25) # σ AIRS\n\n    best = (-1.0, 1.0, 1.0, 1.0, 1.0)  # (GLL, cf, ca, mf, ma)\n    total = len(grid_cf) * len(grid_ca) * len(grid_mf) * len(grid_ma)\n    with tqdm(total=total, desc=\"Tune (cf,ca,mf,ma)\", dynamic_ncols=True) as pbar:\n        for cf, ca, mf, ma in product(grid_cf, grid_ca, grid_mf, grid_ma):\n            g = _eval_gll_with_params(y_sub, mu_sub, s_sub, naive_mean, naive_sigma, cf, ca, mf, ma)\n            if g > best[0]:\n                best = (g, cf, ca, mf, ma)\n                pbar.set_postfix(best=f\"{best[0]:.6f}\",\n                                 cf=f\"{best[1]:.3f}\", ca=f\"{best[2]:.3f}\",\n                                 mf=f\"{best[3]:.3f}\", ma=f\"{best[4]:.3f}\")\n            pbar.update(1)\n\n    # 원본 train 전체에서 검증\n    g_full = _eval_gll_with_params(y_true, mu, sigma, naive_mean, naive_sigma,\n                                   best[1], best[2], best[3], best[4])\n    log(f\"[TUNE] coarse best on FULL: GLL={g_full:.6f}  \"\n        f\"(cf={best[1]:.3f}, ca={best[2]:.3f}, mf={best[3]:.3f}, ma={best[4]:.3f})\")\n\n    if not refine:\n        return g_full, best[1], best[2], best[3], best[4]\n\n    # --------- 근방 재탐색(로컬 리파인) ---------\n    def _around(x, width=0.08, k=7):\n        lo = max(0.10, x * (1 - width))\n        hi = x * (1 + width)\n        return np.linspace(lo, hi, k)\n\n    cf_cand = _around(best[1], width=0.05, k=7)\n    ca_cand = _around(best[2], width=0.05, k=7)\n    mf_cand = _around(best[3], width=0.10, k=9)\n    ma_cand = _around(best[4], width=0.10, k=9)\n\n    best_ref = (g_full, best[1], best[2], best[3], best[4])\n    total2 = len(cf_cand) * len(ca_cand) * len(mf_cand) * len(ma_cand)\n    with tqdm(total=total2, desc=\"Refine (cf,ca,mf,ma)\", dynamic_ncols=True) as pbar:\n        for cf, ca, mf, ma in product(cf_cand, ca_cand, mf_cand, ma_cand):\n            g = _eval_gll_with_params(y_true, mu, sigma, naive_mean, naive_sigma, cf, ca, mf, ma)\n            if g > best_ref[0]:\n                best_ref = (g, cf, ca, mf, ma)\n                pbar.set_postfix(best=f\"{best_ref[0]:.6f}\",\n                                 cf=f\"{best_ref[1]:.3f}\", ca=f\"{best_ref[2]:.3f}\",\n                                 mf=f\"{best_ref[3]:.3f}\", ma=f\"{best_ref[4]:.3f}\")\n            pbar.update(1)\n\n    log(f\"[TUNE] refined best: GLL={best_ref[0]:.6f}  \"\n        f\"(cf={best_ref[1]:.3f}, ca={best_ref[2]:.3f}, mf={best_ref[3]:.3f}, ma={best_ref[4]:.3f})\")\n    return best_ref\n\n# -------------------- Main -------------------- #\ncfg = Config()\nlog(\"==== Ariel 2025 pipeline start ====\")\n\n# 모델 로드\nfgs_model, airs_model = load_models()\n\n# 디버그/전체 설정\nfrac = DEBUG_FRAC if DEBUG_MODE else 1.0\nseed = DEBUG_SEED\n\n# 1) TRAIN\nlog(\"==> Stage 1: TRAIN pipeline\")\nmu_tr, sigma_tr, ids_tr, star_tr, pre_tr = run_pipeline(\"train\", cfg, fgs_model, airs_model, frac=frac, seed=seed)\n\n# Train GT\ntrain_df = pd.read_csv(f\"{ROOT_PATH}/train.csv\")\nWL_COLS = sorted([c for c in train_df.columns if c.startswith(\"wl_\") or c.startswith(\"wavelength_\")],\n                 key=lambda x: int(\"\".join([d for d in x if d.isdigit()])))\ny_true_tr = train_df.set_index(\"planet_id\").loc[ids_tr, WL_COLS].values.astype(np.float64)\n\n# naive 기준\narr = y_true_tr.astype(np.float64)\nNAIVE_MEAN  = float(np.nanmean(arr))\nNAIVE_SIGMA = float(np.nanstd(arr) + 1e-12)\n\ngll_base = gll_score_numpy(y_true_tr, mu_tr, sigma_tr, NAIVE_MEAN, NAIVE_SIGMA)\nlog(f\"[TRAIN] baseline GLL = {gll_base:.6f}\")\n\n# σ 튠\nlog(\"==> Stage 2: (μ,σ) joint tuning on TRAIN\")\nif DEBUG_MODE:\n    grid_cf = np.linspace(0.95, 1.05, 5)   # 좁은 μ 범위\n    grid_ca = np.linspace(0.95, 1.05, 5)\n    grid_mf = np.linspace(0.8,  1.4,  9)   # DEBUG에서 가벼운 σ 범위\n    grid_ma = np.linspace(0.8,  1.4,  9)\n    max_eval_planets = 120\nelse:\n    grid_cf = np.linspace(0.90, 1.10, 9)\n    grid_ca = np.linspace(0.90, 1.10, 9)\n    grid_mf = np.linspace(0.6,  1.8,  25)\n    grid_ma = np.linspace(0.6,  1.8,  25)\n    max_eval_planets =1100\n\nbest_gll, cf_best, ca_best, mf_best, ma_best = tune_mu_sigma_on_train(\n    y_true_tr, mu_tr, sigma_tr, NAIVE_MEAN, NAIVE_SIGMA,\n    grid_cf=grid_cf, grid_ca=grid_ca, grid_mf=grid_mf, grid_ma=grid_ma,\n    seed=DEBUG_SEED, max_eval_planets=max_eval_planets, refine=True\n)\n\nlog(f\"[TRAIN] tuned (μ,σ) ⇒ \"\n    f\"cf={cf_best:.3f}, ca={ca_best:.3f}, mf={mf_best:.3f}, ma={ma_best:.3f} | GLL={best_gll:.6f}\")\n\n# (기존) TEST에서 σ만 곱하던 부분도 아래처럼 μ·σ 둘 다 적용\nlog(\"==> Stage 3: TEST pipeline (apply tuned params)\")\nmu_te, sigma_te, ids_te, star_te, pre_te = run_pipeline(\"test\", cfg, fgs_model, airs_model,\n                                                        frac=DEBUG_FRAC if DEBUG_MODE else 1.0,\n                                                        seed=DEBUG_SEED)\nmu_te, sigma_te = apply_params(mu_te, sigma_te,\n                               cf=cf_best, ca=ca_best, mf=mf_best, ma=ma_best)\nlog(\"✔ applied tuned (μ,σ) to TEST\")\n\n# 제출 파일 만들기\nsample_sub = pd.read_csv(f\"{ROOT_PATH}/sample_submission.csv\", index_col=\"planet_id\")\nmu_df  = pd.DataFrame(mu_te, index=ids_te, columns=WL_COLS)\nsig_df = pd.DataFrame(sigma_te, index=ids_te, columns=[f\"sigma_{i+1}\" for i in range(mu_te.shape[1])])\nsub_dbg = pd.concat([mu_df, sig_df], axis=1)\n\nif DEBUG_MODE:\n    sub_dbg.to_csv(\"submission_debug.csv\")\n    log(f\"[SAVE] submission_debug.csv  shape={sub_dbg.shape}  (rows={len(ids_te)})\")\nelse:\n    sub = sub_dbg.reindex(sample_sub.index)           # 행 정렬\n    sub = sub[sample_sub.columns]                     # 열 정렬\n    sub.to_csv(\"submission.csv\")\n    log(f\"[SAVE] submission.csv  shape={sub.shape}\")\n\nlog(\"==== Done ====\")\n","metadata":{"papermill":{"duration":5.564138,"end_time":"2025-09-19T07:12:01.438184","exception":false,"start_time":"2025-09-19T07:11:55.874046","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T16:02:51.686189Z","iopub.execute_input":"2025-09-24T16:02:51.686597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# Bias tuning (z-hat via Ridge) + α search + Viz  [AFTER를 베이스라인으로 채택]\n# ============================================\n\nimport numpy as np, pandas as pd, re, matplotlib.pyplot as plt\n\n# ---------- helpers ----------\ndef _wavelength_positions(cols):\n    xs = []\n    for c in cols:\n        m = re.findall(r'\\d+', str(c))\n        xs.append(int(m[-1]) if m else None)\n    return np.array(xs) if all(x is not None for x in xs) else np.arange(1, len(cols) + 1)\n\n# gll_score_numpy (필요시 정의)\nif 'gll_score_numpy' not in globals():\n    try:\n        from scipy.stats import norm\n        def gll_score_numpy(y_true, y_pred, sigma_pred,\n                            naive_mean, naive_sigma,\n                            fgs_sigma_true=1e-6, airs_sigma_true=1e-5, fgs_weight=0.4):\n            sigma_pred = np.clip(sigma_pred, 1e-15, None)\n            n, L = sigma_pred.shape\n            sigma_true = np.append([fgs_sigma_true], np.full(L-1, airs_sigma_true))\n            sigma_true = np.tile(sigma_true, (n, 1))\n            weights = np.append([fgs_weight], np.ones(L-1))\n            weights = np.tile(weights, (n, 1))\n            gll_pred  = norm.logpdf(y_true, loc=y_pred, scale=sigma_pred)\n            gll_true  = norm.logpdf(y_true, loc=y_true, scale=sigma_true)\n            gll_naive = norm.logpdf(y_true, loc=naive_mean, scale=naive_sigma)\n            ind = (gll_pred - gll_naive) / (gll_true - gll_naive + 1e-12)\n            return float(np.clip(np.average(ind, weights=weights), 0.0, 1.0))\n    except Exception:\n        # 간단 대체(정규근사 logpdf)\n        def _logpdf(x, mu, sig):\n            sig = np.clip(sig, 1e-15, None)\n            z = (x - mu) / sig\n            return -0.5*np.log(2*np.pi) - np.log(sig) - 0.5*z*z\n        def gll_score_numpy(y_true, y_pred, sigma_pred,\n                            naive_mean, naive_sigma,\n                            fgs_sigma_true=1e-6, airs_sigma_true=1e-5, fgs_weight=0.4):\n            sigma_pred = np.clip(sigma_pred, 1e-15, None)\n            n, L = sigma_pred.shape\n            sigma_true = np.append([fgs_sigma_true], np.full(L-1, airs_sigma_true))\n            sigma_true = np.tile(sigma_true, (n, 1))\n            weights = np.append([fgs_weight], np.ones(L-1))\n            weights = np.tile(weights, (n, 1))\n            gll_pred  = _logpdf(y_true, y_pred, sigma_pred)\n            gll_true  = _logpdf(y_true, y_true, sigma_true)\n            gll_naive = _logpdf(y_true, naive_mean, naive_sigma)\n            ind = (gll_pred - gll_naive) / (gll_true - gll_naive + 1e-12)\n            return float(np.clip(np.average(ind, weights=weights), 0.0, 1.0))\n\n# apply_params (필요시 정의)\nif 'apply_params' not in globals():\n    def apply_params(mu, sigma, cf=1.0, ca=1.0, mf=1.0, ma=1.0):\n        mu2 = mu.copy()\n        mu2[:, 0]  = np.clip(cf * mu2[:, 0],  0.0, None)\n        mu2[:, 1:] = np.clip(ca * mu2[:, 1:], 0.0, None)\n        s2 = sigma.copy()\n        s2[:, 0]  = np.clip(mf * s2[:, 0],  1e-12, None)\n        s2[:, 1:] = np.clip(ma * s2[:, 1:], 1e-12, None)\n        return mu2, s2\n\n# Ridge (필요시 정의)\nif 'fit_ridge' not in globals():\n    def fit_ridge(X, y, lam=1e-2):\n        X = np.asarray(X, float); y = np.asarray(y, float)\n        mu = X.mean(axis=0); sd = X.std(axis=0) + 1e-12\n        Xs = (X - mu) / sd\n        Phi = np.c_[np.ones(len(Xs)), Xs]       # bias 포함\n        I = np.eye(Phi.shape[1]); I[0,0] = 0.0  # bias에는 페널티 X\n        w = np.linalg.solve(Phi.T @ Phi + lam * I, Phi.T @ y)\n        return {\"mu\": mu, \"sd\": sd, \"w\": w}\n\nif 'predict_z' not in globals():\n    def predict_z(model, X):\n        X = np.asarray(X, float)\n        Xs = (X - model[\"mu\"]) / model[\"sd\"]\n        Phi = np.c_[np.ones(len(Xs)), Xs]\n        return Phi @ model[\"w\"]\n\n# 행성별 편향 보정\nif 'apply_planetwise_bias' not in globals():\n    def apply_planetwise_bias(mu, sigma, s, alpha_mu=0.4, alpha_sig=0.10, z_cap=0.6):\n        s = np.clip(np.asarray(s, float), -float(z_cap), float(z_cap)).reshape(-1, 1)\n        mu2  = np.clip(mu + alpha_mu * s * sigma, 0.0, None)\n        sig2 = np.clip(sigma * (1.0 + alpha_sig * np.abs(s)), 1e-12, None)\n        return mu2, sig2\n\nif 'tune_alpha_on_train' not in globals():\n    def tune_alpha_on_train(y_true, mu, sigma, s, naive_mean, naive_sigma,\n                            grid_alpha_mu=None, grid_alpha_sig=None, z_cap=0.6):\n        if grid_alpha_mu is None:  grid_alpha_mu  = np.linspace(0.00, 0.80, 9)\n        if grid_alpha_sig is None: grid_alpha_sig = np.linspace(0.00, 0.30, 7)\n        best = (-1.0, 0.0, 0.0)\n        for am in grid_alpha_mu:\n            for asg in grid_alpha_sig:\n                mu2, sg2 = apply_planetwise_bias(mu, sigma, s, alpha_mu=am, alpha_sig=asg, z_cap=z_cap)\n                g = gll_score_numpy(y_true, mu2, sg2, naive_mean, naive_sigma)\n                if g > best[0]:\n                    best = (g, am, asg)\n        return best\n\n# ---------- required core objects check ----------\nfor name in ['mu_tr', 'sigma_tr', 'y_true_tr', 'ids_tr']:\n    if name not in globals():\n        raise RuntimeError(f\"[need] '{name}'가 세션에 필요합니다. (Train 파이프라인 먼저 실행)\")\n\n# WL_COLS / xlam\nif 'WL_COLS' in globals():\n    xlam = _wavelength_positions(WL_COLS).astype(float)\nelse:\n    xlam = np.arange(1, mu_tr.shape[1] + 1, dtype=float)\n\n# naive 기준\nif 'NAIVE_MEAN' not in globals() or 'NAIVE_SIGMA' not in globals():\n    arr = np.asarray(y_true_tr, float)\n    NAIVE_MEAN  = float(np.nanmean(arr))\n    NAIVE_SIGMA = float(np.nanstd(arr) + 1e-12)\n\n# 튠 파라미터 (없으면 1.0)\ncf_best = globals().get('cf_best', 1.0)\nca_best = globals().get('ca_best', 1.0)\nmf_best = globals().get('mf_best', 1.0)\nma_best = globals().get('ma_best', 1.0)\n\n# ---- 1) (cf,ca,mf,ma) 적용본 만들기 (Train/Test) ----\nmu_tr_adj, sigma_tr_adj = apply_params(mu_tr, sigma_tr, cf=cf_best, ca=ca_best, mf=mf_best, ma=ma_best)\nif 'mu_te' in globals() and 'sigma_te' in globals():\n    mu_te_adj, sigma_te_adj = apply_params(mu_te, sigma_te, cf=cf_best, ca=ca_best, mf=mf_best, ma=ma_best)\nelse:\n    mu_te_adj = sigma_te_adj = None\n\n# ---- 2) 피처 테이블 (있으면 고급, 없으면 간소 대체) ----\nfeat_names = [\"num_dip_periods\", \"shape_vu_ratio\", \"centroid_x_std\", \"centroid_speed_mean\"]\n_use_simple_feat = True\nif 'build_fgs1_feature_table' in globals() and 'cfg' in globals():\n    try:\n        feat_tr = build_fgs1_feature_table(cfg, ids_tr, split=\"train\")[feat_names].astype(float)\n        feat_tr = (feat_tr\n                   .replace([np.inf, -np.inf], np.nan)\n                   .fillna(feat_tr.median(numeric_only=True))\n                   .reindex(ids_tr))\n        _use_simple_feat = False\n    except Exception as e:\n        print(f\"[feat] high-level feature failed → fallback: {e}\")\n\nif _use_simple_feat:\n    eps = 1e-12\n    Ztr_tmp = (y_true_tr - mu_tr_adj) / (sigma_tr_adj + eps)\n    feat_tr = pd.DataFrame(index=range(len(Ztr_tmp)))\n    feat_tr[\"res_std\"]   = np.nanstd(Ztr_tmp, axis=1)\n    feat_tr[\"frac_gt1\"]  = np.mean(np.abs(Ztr_tmp) > 1.0, axis=1)\n    feat_tr[\"frac_gt2\"]  = np.mean(np.abs(Ztr_tmp) > 2.0, axis=1)\n    feat_tr[\"res_mean\"]  = np.nanmean(Ztr_tmp, axis=1)\n    feat_names = list(feat_tr.columns)\n\n# ---- 3) z_target & ẑ 회귀 ----\neps = 1e-12\nZtr = (y_true_tr - mu_tr_adj) / (sigma_tr_adj + eps)\nz_target = np.nanmedian(Ztr, axis=1)\nz_model  = fit_ridge(feat_tr.values, z_target, lam=1e-2)\nz_hat_tr = predict_z(z_model, feat_tr.values)\n\n# ---- 4) α(αμ, ασ) 그리드 서치 + 적용 ----\nbest_gll_tr, a_mu, a_sig = tune_alpha_on_train(\n    y_true_tr, mu_tr_adj, sigma_tr_adj, z_hat_tr,\n    NAIVE_MEAN, NAIVE_SIGMA,\n    grid_alpha_mu=np.linspace(0.00, 2.0, 21),\n    grid_alpha_sig=np.linspace(0.00, 0.20, 21),\n    z_cap=0.6\n)\n\n# --- (A) 비포 스냅샷 저장 (이번 런의 '현재' 상태) ---\nmu_tr_before = mu_tr_adj.copy()\nsigma_tr_before = sigma_tr_adj.copy()\n\n# 보정후(바이어스 적용)\nmu_tr_bias, sigma_tr_bias = apply_planetwise_bias(\n    mu_tr_adj, sigma_tr_adj, z_hat_tr,\n    alpha_mu=a_mu, alpha_sig=a_sig, z_cap=0.6\n)\nprint(f\"[BIAS] αμ={a_mu:.3f}, ασ={a_sig:.3f}  → GLL(train)={best_gll_tr:.6f}\")\n\n# --- (B) 보정후를 베이스라인으로 채택 ---\nmu_tr_adj, sigma_tr_adj = mu_tr_bias, sigma_tr_bias\n\n# Test에도 동일 모델/α 적용(있을 때)\nif mu_te_adj is not None and 'ids_te' in globals():\n    try:\n        if not _use_simple_feat and 'build_fgs1_feature_table' in globals():\n            feat_te = build_fgs1_feature_table(cfg, ids_te, split=\"test\")[feat_names].astype(float)\n            feat_te = (feat_te\n                       .replace([np.inf, -np.inf], np.nan)\n                       .fillna(feat_te.median(numeric_only=True))\n                       .reindex(ids_te))\n        else:\n            # 간소 대체: Train의 중앙값 특징을 복사(보수적)\n            feat_te = pd.DataFrame(\n                np.tile(feat_tr.median(numeric_only=True).values, (len(ids_te), 1)),\n                index=ids_te, columns=feat_names\n            )\n\n        z_hat_te = predict_z(z_model, feat_te.values)\n        mu_te_bias, sigma_te_bias = apply_planetwise_bias(\n            mu_te_adj, sigma_te_adj, z_hat_te,\n            alpha_mu=a_mu, alpha_sig=a_sig, z_cap=0.6\n        )\n        # 이후 파이프라인도 보정후 사용\n        mu_te, sigma_te = mu_te_bias, sigma_te_bias\n        print(f\"[BIAS-TEST] applied αμ, ασ to TEST (N={len(ids_te)})\")\n    except Exception as e:\n        print(f\"[BIAS-TEST] skipped: {e}\")\n\n# ---- 5) Viz: BEFORE vs AFTER (Train) ----\ndef _cov_at(Z, thr=1.0): return float(np.mean(np.abs(Z) <= thr))\n\nZ_before = (y_true_tr - mu_tr_before) / (sigma_tr_before + eps)   # 스냅샷(비포)\nZ_after  = (y_true_tr - mu_tr_adj)    / (sigma_tr_adj    + eps)   # 채택된 보정후\n\ngll_b = float(gll_score_numpy(y_true_tr, mu_tr_before, sigma_tr_before, NAIVE_MEAN, NAIVE_SIGMA))\ngll_a = float(gll_score_numpy(y_true_tr, mu_tr_adj,    sigma_tr_adj,    NAIVE_MEAN, NAIVE_SIGMA))\n\nprint(\n    f\"[BIAS/VIZ] GLL {gll_b:.6f} → {gll_a:.6f} | \"\n    f\"cov@1σ { _cov_at(Z_before,1):.4f} → { _cov_at(Z_after,1):.4f} | \"\n    f\"cov@2σ { _cov_at(Z_before,2):.4f} → { _cov_at(Z_after,2):.4f}\"\n)\n\n# Z 히스토그램\nplt.figure(figsize=(6.8,4.2))\nplt.hist(Z_before.ravel(), bins=60, density=True, alpha=0.45, label=\"before\")\nplt.hist(Z_after.ravel(),  bins=60, density=True, alpha=0.45, label=\"after\")\nxs = np.linspace(-5, 5, 401)\npdf = (1/np.sqrt(2*np.pi))*np.exp(-0.5*xs**2)\nplt.plot(xs, pdf, lw=1.5, label=\"N(0,1)\")\nplt.title(\"Residual Z (Train): before vs after bias-tuning\")\nplt.xlabel(\"Z\"); plt.ylabel(\"density\"); plt.legend()\nplt.tight_layout(); plt.show()\n\n# 무작위 스펙트럼 before/after\ndef plot_train_spectra_before_after(n_samples=18, seed=7):\n    rng = np.random.default_rng(seed)\n    idx_all = np.arange(len(ids_tr))\n    pick = rng.choice(idx_all, size=min(n_samples, len(idx_all)), replace=False)\n    rows = int(np.ceil(len(pick) / 3)); cols = int(min(3, len(pick)))\n    plt.figure(figsize=(5*cols, 3.6*rows))\n    for k, i in enumerate(pick, 1):\n        yt = y_true_tr[i]\n        mp0, sp0 = mu_tr_before[i], sigma_tr_before[i]   # BEFORE\n        mp1, sp1 = mu_tr_adj[i],    sigma_tr_adj[i]      # AFTER(채택)\n        ax = plt.subplot(rows, cols, k)\n        ax.plot(xlam, yt,  lw=1.1, label=\"truth\")\n        ax.plot(xlam, mp0, lw=1.1, label=\"pred (before)\")\n        ax.fill_between(xlam, mp0-sp0, mp0+sp0, alpha=0.15, linewidth=0)\n        ax.plot(xlam, mp1, lw=1.1, label=\"pred (after)\")\n        ax.fill_between(xlam, mp1-sp1, mp1+sp1, alpha=0.15, linewidth=0)\n        ax.set_title(f\"planet_id={ids_tr[i]}\")\n        if k == 1: ax.legend()\n    plt.tight_layout(); plt.show()\n\nplot_train_spectra_before_after(n_samples=18)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Integrated: 3-way split → FGS1 mini features & corr\n#             → Ridge z-hat (FGS1 + StarInfo) + (αμ,ασ) grid-search conditional bias\n#             → AFTER viz (+optional BEFORE overlay)\n#             → σ-global (k) tune → submission\n#   - AFTER(보정후)가 현재 베이스라인(mu_tr_adj/sigma_tr_adj)\n#   - z_target은 가능하면 BEFORE 스냅샷으로 계산\n#   - z_th = 0.6 고정, |ẑ| ≥ z_th 인 행성만 보정\n#   - αμ ∈ [0,2], ασ ∈ [0,0.20] 그리드서치\n#   - NEW1: StarInfo도 ẑ 학습 피처에 포함\n#   - NEW2: [선형성] train true 분산 하위 30% → 선형 라벨, StarInfo로 분류 학습\n#           test에서 선형 예측 시 mu를 직선화(linearize) 보정\n# ============================================================\n\nimport numpy as np, pandas as pd, matplotlib.pyplot as plt, re, os\nfrom tqdm.auto import tqdm\n\n# ---------------- safety: display fallback ----------------\ntry:\n    from IPython.display import display\nexcept Exception:\n    def display(x):\n        try:\n            print(x.to_string())\n        except Exception:\n            print(x)\n\n# ---------------- helpers ----------------\ndef _wavelength_positions(cols):\n    xs = []\n    for c in cols:\n        m = re.findall(r'\\d+', str(c))\n        xs.append(int(m[-1]) if m else None)\n    return np.array(xs) if all(x is not None for x in xs) else np.arange(1, len(cols) + 1)\n\n# gll_score_numpy (fallback if missing)\nif 'gll_score_numpy' not in globals():\n    try:\n        from scipy.stats import norm\n        def gll_score_numpy(y_true, y_pred, sigma_pred,\n                            naive_mean, naive_sigma,\n                            fgs_sigma_true=1e-6, airs_sigma_true=1e-5, fgs_weight=0.4):\n            sigma_pred = np.clip(sigma_pred, 1e-15, None)\n            n, L = sigma_pred.shape\n            sigma_true = np.append([fgs_sigma_true], np.full(L-1, airs_sigma_true))\n            sigma_true = np.tile(sigma_true, (n, 1))\n            weights = np.append([fgs_weight], np.ones(L-1))\n            weights = np.tile(weights, (n, 1))\n            gll_pred  = norm.logpdf(y_true, loc=y_pred, scale=sigma_pred)\n            gll_true  = norm.logpdf(y_true, loc=y_true, scale=sigma_true)\n            gll_naive = norm.logpdf(y_true, loc=naive_mean, scale=naive_sigma)\n            ind = (gll_pred - gll_naive) / (gll_true - gll_naive + 1e-12)\n            return float(np.clip(np.average(ind, weights=weights), 0.0, 1.0))\n    except Exception:\n        def _logpdf(x, mu, sig):\n            sig = np.clip(sig, 1e-15, None)\n            z = (x - mu) / sig\n            return -0.5*np.log(2*np.pi) - np.log(sig) - 0.5*z*z\n        def gll_score_numpy(y_true, y_pred, sigma_pred,\n                            naive_mean, naive_sigma,\n                            fgs_sigma_true=1e-6, airs_sigma_true=1e-5, fgs_weight=0.4):\n            sigma_pred = np.clip(sigma_pred, 1e-15, None)\n            n, L = sigma_pred.shape\n            sigma_true = np.append([fgs_sigma_true], np.full(L-1, airs_sigma_true))\n            sigma_true = np.tile(sigma_true, (n, 1))\n            weights = np.append([fgs_weight], np.ones(L-1))\n            weights = np.tile(weights, (n, 1))\n            gll_pred  = _logpdf(y_true, y_pred, sigma_pred)\n            gll_true  = _logpdf(y_true, y_true, sigma_true)\n            gll_naive = _logpdf(y_true, naive_mean, naive_sigma)\n            ind = (gll_pred - gll_naive) / (gll_true - gll_naive + 1e-12)\n            return float(np.clip(np.average(ind, weights=weights), 0.0, 1.0))\n\n# ---------------- required objects ----------------\nfor name in ['mu_tr_adj', 'sigma_tr_adj', 'y_true_tr', 'ids_tr']:\n    assert name in globals(), f\"[need] '{name}' 가 세션에 필요합니다.\"\n\n# (옵션) 이전 런 BEFORE 스냅샷(있으면 라벨링/비교용)\nmu_tr_before    = globals().get('mu_tr_before', None)\nsigma_tr_before = globals().get('sigma_tr_before', None)\n_have_before = (mu_tr_before is not None) and (sigma_tr_before is not None)\n\n# xlam (파장축)\nif 'WL_COLS' in globals():\n    xlam = _wavelength_positions(WL_COLS).astype(float)\nelse:\n    xlam = np.arange(1, np.asarray(mu_tr_adj).shape[1] + 1, dtype=float)\n\n# naive 기준(없으면 계산)\nif 'NAIVE_MEAN' not in globals() or 'NAIVE_SIGMA' not in globals():\n    arr = np.asarray(y_true_tr, float)\n    NAIVE_MEAN  = float(np.nanmean(arr))\n    NAIVE_SIGMA = float(np.nanstd(arr) + 1e-12)\n\n# ============================================================\n# 3-way split (AFTER 기준, viz용)\n# ============================================================\neps  = 1e-12\nz_th = 0.2   # ← 고정\n\nZ_after_full = (y_true_tr - mu_tr_adj) / (sigma_tr_adj + eps)\nz_med_after  = np.nanmedian(Z_after_full, axis=1)\nz_mean_after = np.nanmean(Z_after_full, axis=1)\nfrac_hi = np.mean(Z_after_full >  +1.0, axis=1)\nfrac_lo = np.mean(Z_after_full <  -1.0, axis=1)\n\ndef _corr(a, b):\n    a = a - np.nanmean(a); b = b - np.nanmean(b)\n    na = np.linalg.norm(a); nb = np.linalg.norm(b)\n    if na*nb == 0 or ~np.isfinite([na, nb]).all(): return np.nan\n    return float((a@b) / (na*nb))\n\ncorr = np.array([_corr(y_true_tr[i], mu_tr_adj[i]) for i in range(len(ids_tr))])\n\nmask_over  = (z_med_after < -z_th)     # μ>y\nmask_under = (z_med_after >  +z_th)    # μ<y\nmask_well  = ~(mask_over | mask_under)\nmask_well &= (np.nan_to_num(corr, nan=0.0) >= 0.10)\n\nqc3 = pd.DataFrame({\n    \"planet_id\": ids_tr,\n    \"z_median\":  z_med_after,\n    \"z_mean\":    z_mean_after,\n    \"frac_gt1σ\": frac_hi,\n    \"frac_lt-1σ\":frac_lo,\n    \"corr\":      corr,\n    \"category\":  np.where(mask_over,  \"over(μ>y)\",\n                   np.where(mask_under, \"under(μ<y)\", \"well\"))\n}).set_index(\"planet_id\")\n\nwell_ids  = qc3.index[qc3[\"category\"] == \"well\"].tolist()\nover_ids  = qc3.index[qc3[\"category\"] == \"over(μ>y)\"].tolist()\nunder_ids = qc3.index[qc3[\"category\"] == \"under(μ<y)\"].tolist()\n\nprint(f\"[SPLIT-3WAY(AFTER)] well={len(well_ids)}  over={len(over_ids)}  under={len(under_ids)} / total={len(ids_tr)}\")\ndisplay(qc3.reindex(qc3['z_median'].abs().sort_values(ascending=False).index).head(5))\n\n# ============================================================\n# FGS1 features (4개)\n# ============================================================\ndef _safe_median_baseline(x, win=5000):\n    n = len(x); w = min(int(win), max(11, n//10))\n    s = pd.Series(x)\n    return s.rolling(w, center=True).median().bfill().ffill().to_numpy()\n\ndef extract_transit_detection_features(total_flux):\n    x = np.asarray(total_flux, dtype=np.float64)\n    if x.size == 0: return {\"num_dip_periods\": 0}\n    thr = float(np.mean(x) - 1.0*np.std(x))\n    below = x < thr\n    if not np.any(below): return {\"num_dip_periods\": 0}\n    d = np.diff(np.concatenate(([False], below, [False])).astype(int))\n    starts = np.where(d == 1)[0]; ends = np.where(d == -1)[0]\n    durations = ends - starts\n    return {\"num_dip_periods\": int(len(durations))}\n\ndef extract_transit_shape_features(total_flux, baseline_win=5000, local_win=600):\n    x = np.asarray(total_flux, dtype=np.float64); n = len(x)\n    if n < 10: return {\"shape_vu_ratio\": 0.0}\n    base = _safe_median_baseline(x, baseline_win)\n    d = x - base\n    idx0 = int(np.argmin(d)); depth = float(-d[idx0])\n    if not np.isfinite(depth) or depth <= 0: return {\"shape_vu_ratio\": 0.0}\n    w = int(min(local_win, n-1)); lo, hi = max(0, idx0 - w//2), min(n, idx0 + w//2)\n    local = d[lo:hi]\n    if local.size < 5: return {\"shape_vu_ratio\": 0.0}\n    d1 = np.diff(local)\n    vu = float(np.median(np.abs(d1)) / (depth + 1e-12))\n    return {\"shape_vu_ratio\": vu}\n\ndef extract_centroid_track_features(signal_3d, step=50, max_frames=20000):\n    arr = np.asarray(signal_3d, dtype=np.float32)\n    if arr.ndim != 3 or arr.shape[0] == 0:\n        return {\"centroid_x_std\": 0.0, \"centroid_speed_mean\": 0.0}\n    T, H, W = arr.shape\n    idx = np.arange(0, T, step, dtype=int)\n    if idx.size > max_frames: idx = idx[:max_frames]\n    xs = np.arange(W, dtype=np.float32)\n    cx = []\n    for t in idx:\n        frame = arr[t]; s = float(frame.sum())\n        if s <= 0:\n            cx.append((cx[-1] if cx else W/2.0))\n        else:\n            cx.append(float((frame * xs[None, :]).sum() / s))\n    cx = np.asarray(cx)\n    if cx.size < 3:\n        return {\"centroid_x_std\": 0.0, \"centroid_speed_mean\": 0.0}\n    vx = np.diff(cx); speed = np.abs(vx)\n    return {\"centroid_x_std\": float(np.std(cx)), \"centroid_speed_mean\": float(np.mean(speed))}\n\ndef build_fgs1_feature_table(cfg, ids, split=\"train\", step_for_centroid=50):\n    assert 'SignalProcessor' in globals(), \"[need] SignalProcessor 클래스가 필요합니다.\"\n    sp = SignalProcessor(cfg, split)\n    rows = []\n    for pid in tqdm(ids, desc=f\"FGS1 feature extract ({split})\", dynamic_ncols=True):\n        sig = sp._calibrate_single_signal(pid, \"FGS1\")  # (T,H,W)\n        total_flux = sig.sum(axis=(1,2))\n        out = {}\n        out.update(extract_transit_detection_features(total_flux))\n        out.update(extract_transit_shape_features(total_flux))\n        out.update(extract_centroid_track_features(sig, step=step_for_centroid))\n        rows.append(pd.Series(out, name=pid))\n    return pd.DataFrame(rows)\n\n# ============================================================\n# Star info feature builder\n# ============================================================\nSTAR_TRAIN_PATH = \"/kaggle/input/ariel-data-challenge-2025/train_star_info.csv\"\nSTAR_TEST_PATH  = \"/kaggle/input/ariel-data-challenge-2025/test_star_info.csv\"\n\ndef build_star_feature_table(path_csv, ids, topk_cat=8, train_cols=None, train_medians=None):\n    if not os.path.exists(path_csv):\n        empty = pd.DataFrame(index=ids)\n        if train_cols is not None:\n            return empty.reindex(columns=train_cols).fillna(0.0)\n        return empty\n    df = pd.read_csv(path_csv)\n    key = None\n    for cand in [\"planet_id\", \"Planet_ID\", \"planetId\", \"id\"]:\n        if cand in df.columns:\n            key = cand; break\n    assert key is not None, \"[star] planet_id 키 컬럼이 필요합니다.\"\n    df = df.set_index(key).reindex(ids)\n\n    drop_id_like = [c for c in df.columns if re.search(r\"id$\", c, flags=re.I) or re.search(r\"_id\", c, flags=re.I)]\n    df = df.drop(columns=drop_id_like, errors=\"ignore\")\n\n    num_cols = [c for c in df.columns if pd.api.types.is_numeric_dtype(df[c])]\n    cat_cols = [c for c in df.columns if df[c].dtype == \"object\" or str(df[c].dtype) == \"category\"]\n\n    num_df = df[num_cols].copy()\n    cat_df_list = []\n    for c in cat_cols:\n        vc = df[c].astype(\"string\").fillna(\"NA\").value_counts()\n        keep = vc.index[:topk_cat].tolist()\n        tmp = pd.get_dummies(df[c].astype(\"string\").where(df[c].isin(keep), other=\"OTHER\"), prefix=c)\n        cat_df_list.append(tmp)\n    cat_df = pd.concat(cat_df_list, axis=1) if cat_df_list else pd.DataFrame(index=df.index)\n\n    out = pd.concat([num_df, cat_df], axis=1)\n\n    if train_medians is None:\n        med = out.median(numeric_only=True)\n    else:\n        med = train_medians.reindex(out.columns).fillna(0.0)\n    for c in out.columns:\n        if pd.api.types.is_numeric_dtype(out[c]):\n            out[c] = out[c].astype(float).fillna(float(med.get(c, 0.0)))\n        else:\n            out[c] = out[c].fillna(0.0)\n\n    if train_cols is not None:\n        out = out.reindex(columns=train_cols).fillna(0.0)\n\n    return out\n\n# ---- build FGS1 & Star features (TRAIN)\nfeat_tr_fgs = build_fgs1_feature_table(cfg, ids_tr, split=\"train\")\nfgs_feat_names = [\"num_dip_periods\", \"shape_vu_ratio\", \"centroid_x_std\", \"centroid_speed_mean\"]\nfeat_tr_fgs = (feat_tr_fgs[fgs_feat_names].astype(float)\n               .replace([np.inf, -np.inf], np.nan)\n               .fillna(feat_tr_fgs.median(numeric_only=True)))\n\nstar_tr = build_star_feature_table(STAR_TRAIN_PATH, ids_tr, topk_cat=8)\nprint(f\"[STAR/TRAIN] numeric+onehot cols = {len(star_tr.columns)}\")\n\n# 합치기 (ẑ 학습용 전체 피처)\nfeat_tr_all = pd.concat([feat_tr_fgs, star_tr], axis=1)\nfeat_tr_all = feat_tr_all.replace([np.inf, -np.inf], np.nan)\nfeat_tr_all = feat_tr_all.fillna(feat_tr_all.median(numeric_only=True))\nfeature_cols = feat_tr_all.columns.tolist()\n\n# ============================================================\n# Ridge ẑ 타깃: BEFORE 있으면 BEFORE 잔차 median, 없으면 AFTER\n# ============================================================\nif _have_before:\n    Z_label = (y_true_tr - mu_tr_before) / (sigma_tr_before + eps)\n    label_src = \"BEFORE\"\nelse:\n    Z_label = Z_after_full\n    label_src = \"AFTER\"\nz_target = np.nanmedian(Z_label, axis=1)\nprint(f\"[LABEL-z] source={label_src} | |z|>=0.2: {int(np.sum(np.abs(z_target)>=0.2))}/{len(z_target)}\")\n\n# ---- Ridge helpers ----\nif 'fit_ridge' not in globals():\n    def fit_ridge(X, y, lam=1e-2):\n        X = np.asarray(X, float); y = np.asarray(y, float)\n        mu = X.mean(axis=0); sd = X.std(axis=0) + 1e-12\n        Xs = (X - mu) / sd\n        Phi = np.c_[np.ones(len(Xs)), Xs]\n        I = np.eye(Phi.shape[1]); I[0,0] = 0.0\n        w = np.linalg.solve(Phi.T @ Phi + lam * I, Phi.T @ y)\n        return {\"mu\": mu, \"sd\": sd, \"w\": w, \"cols\": feature_cols}\n\nif 'predict_lin' not in globals():\n    def predict_lin(model, X):\n        X = np.asarray(X, float)\n        Xs = (X - model[\"mu\"]) / model[\"sd\"]\n        Phi = np.c_[np.ones(len(Xs)), Xs]\n        return Phi @ model[\"w\"]\n\ndef apply_planetwise_bias_conditional(mu, sigma, s, alpha_mu=0.8, alpha_sig=0.10, z_cap=0.6):\n    s = np.asarray(s, float).reshape(-1, 1)\n    s = np.where(np.abs(s) >= z_cap, np.clip(s, -z_cap, z_cap), 0.0)\n    mu2  = np.clip(mu + alpha_mu * s * sigma, 0.0, None)\n    sig2 = np.clip(sigma * (1.0 + alpha_sig * np.abs(s)), 1e-12, None)\n    return mu2, sig2\n\n# ============================================================\n# NEW2: \"선형\" 라벨 만들기 (train true 분산 하위 30%)\n#       → StarInfo만으로 릿지 분류(스코어) + 최적 threshold 선택\n# ============================================================\ntrue_var = np.nanvar(np.asarray(y_true_tr, float), axis=1)\nvar_thr  = np.nanquantile(true_var, 0.20)\ny_linear = (true_var <= var_thr).astype(float)\nprint(f\"[LINEAR/LABEL] var_thr(p30)={var_thr:.6e} | linear={int(y_linear.sum())}/{len(y_linear)}\")\n\n# StarInfo로만 학습 (요구사항)\nX_star_tr = star_tr.to_numpy(float)\nlinear_model = fit_ridge(X_star_tr, y_linear, lam=1e-2)\nscore_tr = predict_lin(linear_model, X_star_tr)\n\n# threshold: 단순 그리드에서 정확도 최대값\nths_grid = np.linspace(np.nanmin(score_tr)-1e-6, np.nanmax(score_tr)+1e-6, 301)\nbest_acc, best_th = -1.0, 0.5\nfor th in ths_grid:\n    pred = (score_tr >= th).astype(float)\n    acc  = (pred == y_linear).mean()\n    if acc > best_acc:\n        best_acc, best_th = acc, th\nprint(f\"[LINEAR/CLF] best_th={best_th:.4f}, acc(train)={best_acc:.3f}\")\n\n# ============================================================\n# (1) Train: 전체 피처 → Ridge ẑ\n# ============================================================\nX_tr = feat_tr_all.reindex(ids_tr)[feature_cols].to_numpy(float)\nz_model  = fit_ridge(X_tr, z_target, lam=1e-2)\nz_hat_tr = predict_lin(z_model, X_tr)\n\nm = np.isfinite(z_hat_tr) & np.isfinite(z_target)\ncorr_zhat = float(np.corrcoef(z_hat_tr[m], z_target[m])[0,1]) if m.sum()>2 else np.nan\nprint(f\"[Ridge-z] z_th={z_th:.2f} | corr(ẑ, target)={corr_zhat:+.3f}\")\n\n# (2) α-grid search\ngrid_alpha_mu  = np.linspace(0.00, 2.00, 21)\ngrid_alpha_sig = np.linspace(0.00, 0.20, 21)\n\ng_base = gll_score_numpy(y_true_tr, mu_tr_adj, sigma_tr_adj, NAIVE_MEAN, NAIVE_SIGMA)\nbest = (g_base, 0.0, 0.0)\nbest_mu, best_sg = mu_tr_adj, sigma_tr_adj\n\nfor a_mu in grid_alpha_mu:\n    for a_sig in grid_alpha_sig:\n        mu_tmp, sg_tmp = apply_planetwise_bias_conditional(mu_tr_adj, sigma_tr_adj,\n                                                           z_hat_tr, alpha_mu=a_mu, alpha_sig=a_sig, z_cap=z_th)\n        g = gll_score_numpy(y_true_tr, mu_tmp, sg_tmp, NAIVE_MEAN, NAIVE_SIGMA)\n        if g > best[0]:\n            best = (g, a_mu, a_sig)\n            best_mu, best_sg = mu_tmp, sg_tmp\nprint(f\"[Ridge/GRID] base={g_base:.6f} → best GLL={best[0]:.6f} at αμ={best[1]:.2f}, ασ={best[2]:.2f}\")\n\n# (3) 적용: BEFORE 스냅샷 저장 후 채택\nmu_tr_before, sigma_tr_before = mu_tr_adj.copy(), sigma_tr_adj.copy()\nmu_tr_adj, sigma_tr_adj = best_mu, best_sg\n_have_before = True\n\n# ============================================================\n# Minimal FGS1 feature correlation (표시는 FGS1 4개만 유지)\n# ============================================================\ndf_anal = qc3[[\"z_median\", \"category\"]].join(feat_tr_fgs, how=\"inner\")\nfrom scipy.stats import spearmanr\ncors = []\nfor c in fgs_feat_names:\n    s = df_anal[c].astype(float)\n    m = s.notna() & np.isfinite(s) & np.isfinite(df_anal[\"z_median\"])\n    rho, p = spearmanr(s[m], df_anal.loc[m, \"z_median\"]) if m.sum() >= 8 else (np.nan, np.nan)\n    cors.append((c, rho, p, int(m.sum())))\ncor_df = (pd.DataFrame(cors, columns=[\"feature\",\"spearman_rho\",\"pvalue\",\"n\"])\n            .sort_values(\"spearman_rho\", ascending=False)\n            .reset_index(drop=True))\ndisplay(cor_df)\n\n# ============================================================\n# TEST에도 동일 로직 적용 (FGS1 + StarInfo)\n#   + NEW2: 선형 분류로 선형 예측 시 mu 직선화 보정\n# ============================================================\ndef _linearize_row(mu_row, x, blend=1.0):\n    \"\"\"mu_row를 x(파장축)에 대해 1차 직선으로 적합하고 blend 비율로 덮어쓰기.\"\"\"\n    x = np.asarray(x, float)\n    if x.ndim == 1: X = np.c_[np.ones_like(x), x]\n    else: X = np.c_[np.ones(x.shape[0]), x[:,0]]\n    w, *_ = np.linalg.lstsq(X, mu_row, rcond=None)\n    line = X @ w\n    return np.clip((1.0 - blend) * mu_row + blend * line, 0.0, None)\n\nif all(k in globals() for k in [\"ids_te\",\"mu_te\",\"sigma_te\"]):\n    # FGS1\n    try:\n        feat_te_fgs = build_fgs1_feature_table(cfg, ids_te, split=\"test\")[fgs_feat_names].astype(float)\n        feat_te_fgs = feat_te_fgs.replace([np.inf, -np.inf], np.nan).fillna(feat_te_fgs.median(numeric_only=True))\n    except Exception as e:\n        print(f\"[FGS1-TEST] feature build failed → fallback to train-median: {e}\")\n        feat_te_fgs = pd.DataFrame(np.tile(feat_tr_fgs.median(numeric_only=True).values, (len(ids_te), 1)),\n                                   index=ids_te, columns=fgs_feat_names)\n\n    # Star info (train 스키마에 정렬)\n    star_medians_tr = star_tr.median(numeric_only=True)\n    if os.path.exists(STAR_TEST_PATH):\n        star_te = build_star_feature_table(STAR_TEST_PATH, ids_te, topk_cat=8,\n                                           train_cols=star_tr.columns, train_medians=star_medians_tr)\n    else:\n        star_te = build_star_feature_table(STAR_TRAIN_PATH, ids_te, topk_cat=8,\n                                           train_cols=star_tr.columns, train_medians=star_medians_tr)\n    print(f\"[STAR/TEST] aligned cols = {len(star_te.columns)}\")\n\n    # ẑ 예측용 전체 피처\n    feat_te_all = pd.concat([feat_te_fgs, star_te], axis=1).reindex(columns=feature_cols)\n    med_all_tr = feat_tr_all.median(numeric_only=True)\n    for c in feat_te_all.columns:\n        if pd.api.types.is_numeric_dtype(feat_te_all[c]):\n            feat_te_all[c] = feat_te_all[c].astype(float).fillna(float(med_all_tr.get(c, 0.0)))\n        else:\n            feat_te_all[c] = feat_te_all[c].fillna(0.0)\n\n    X_te = feat_te_all.to_numpy(float)\n    z_hat_te = predict_lin(z_model, X_te)\n\n    # 조건부 바이어스 (ẑ 기반)\n    mu_te, sigma_te = apply_planetwise_bias_conditional(\n        mu_te, sigma_te, z_hat_te, alpha_mu=best[1], alpha_sig=best[2], z_cap=z_th\n    )\n    affected_bias = int(np.sum(np.abs(z_hat_te) >= z_th))\n\n    # NEW2: 선형 분류 → 선형 예측이면 mu 직선화\n    score_te = predict_lin(linear_model, star_te.to_numpy(float))\n    is_linear_te = (score_te >= best_th)\n    n_linear = int(is_linear_te.sum())\n\n    # 직선화 보정 (blend=1.0이면 완전 직선 대체)\n    blend_linear = 1.0\n    x_norm = (xlam - xlam.min()) / (xlam.max() - xlam.min() + 1e-12)\n    for i, flag in enumerate(is_linear_te):\n        if flag:\n            mu_te[i] = _linearize_row(mu_te[i], x_norm, blend=blend_linear)\n\n    print(f\"[Ridge-TEST] conditional-bias affected={affected_bias}/{len(ids_te)} | \"\n          f\"[LINEAR] predicted linear={n_linear}/{len(ids_te)} (blend={blend_linear})\")\nelse:\n    print(\"[Ridge-TEST] skipped (ids_te/mu_te/sigma_te 중 일부 미정의)\")\n\n# ============================================================\n# (옵션) BEFORE vs AFTER: Residual Z & GLL 비교\n# ============================================================\nif _have_before:\n    def _cov_at(Z, thr=1.0): return float(np.mean(np.abs(Z) <= thr))\n    Z_before = (y_true_tr - mu_tr_before) / (sigma_tr_before + eps)\n    Z_after  = (y_true_tr - mu_tr_adj)    / (sigma_tr_adj    + eps)\n    gll_b = float(gll_score_numpy(y_true_tr, mu_tr_before, sigma_tr_before, NAIVE_MEAN, NAIVE_SIGMA))\n    gll_a = float(gll_score_numpy(y_true_tr, mu_tr_adj,    sigma_tr_adj,    NAIVE_MEAN, NAIVE_SIGMA))\n    print(f\"[BIAS/VIZ] GLL {gll_b:.6f} → {gll_a:.6f} | \"\n          f\"cov@1σ {_cov_at(Z_before,1):.4f} → {_cov_at(Z_after,1):.4f} | \"\n          f\"cov@2σ {_cov_at(Z_before,2):.4f} → {_cov_at(Z_after,2):.4f}\")\n\n# ============================================================\n# σ 전역 스케일(k) 튠\n# ============================================================\nmu_curr, sigma_curr = mu_tr_adj, sigma_tr_adj\nk_grid = np.linspace(0.70, 1.30, 31)\nassert any(np.isclose(k_grid, 1.0)), \"[guard] k_grid에 1.0이 포함되어야 합니다.\"\n\ng_base_k = gll_score_numpy(y_true_tr, mu_curr, sigma_curr, NAIVE_MEAN, NAIVE_SIGMA)\nbest_g_k, k_best = g_base_k, 1.0\nfor k in k_grid:\n    g = gll_score_numpy(y_true_tr, mu_curr, sigma_curr * k, NAIVE_MEAN, NAIVE_SIGMA)\n    if g > best_g_k:\n        best_g_k, k_best = g, k\nprint(f\"[σ-GLOBAL] base={g_base_k:.6f}  →  k*={k_best:.3f}  GLL(train,k*)={best_g_k:.6f}\")\n\n# ============================================================\n# Finalize & Save submission\n# ============================================================\nassert 'ids_te' in globals() and 'mu_te' in globals() and 'sigma_te' in globals(), \\\n    \"[need] TEST 예측(mu_te, sigma_te)과 ids_te가 필요합니다.\"\nassert 'ROOT_PATH' in globals(), \"[need] ROOT_PATH 가 필요합니다 (sample_submission.csv 위치).\"\n\nmu_te_final    = np.clip(mu_te, 0.0, None)\nsigma_te_final = np.clip(sigma_te * k_best, 1e-12, None)\n\nsample = pd.read_csv(f\"{ROOT_PATH}/sample_submission.csv\", index_col=\"planet_id\")\nmu_cols  = [c for c in sample.columns if not c.startswith(\"sigma_\")]\nsig_cols = [c for c in sample.columns if     c.startswith(\"sigma_\")]\n\nL = mu_te_final.shape[1]\nmu_cols  = mu_cols[:L]\nsig_cols = sig_cols[:L]\n\nmu_df  = pd.DataFrame(mu_te_final,    index=ids_te, columns=mu_cols)\nsig_df = pd.DataFrame(sigma_te_final, index=ids_te, columns=sig_cols)\nsub_df = pd.concat([mu_df, sig_df], axis=1)\n\nsub = sub_df.reindex(sample.index)[sample.columns]\nif not np.isfinite(sub.values).all():\n    raise ValueError(\"submission에 NaN/Inf가 있습니다. (mu>=0, sigma>=1e-12 확인)\")\n\nsub.to_csv(\"submission.csv\")\nprint(f\"[SAVE] submission.csv  shape={sub.shape}  (rows={len(sub)})\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}