{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.13"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":154281,"sourceType":"competition"},{"sourceId":251736,"sourceType":"datasetVersion"},{"sourceId":18673450,"sourceType":"datasetVersion"},{"sourceId":18673646,"sourceType":"datasetVersion"},{"sourceId":18706996,"sourceType":"datasetVersion"},{"sourceId":18716507,"sourceType":"datasetVersion"},{"sourceId":341776840,"sourceType":"kernelVersion"},{"sourceId":4533,"sourceType":"modelInstanceVersion"}],"isGpuEnabled":false,"isInternetEnabled":true,"language":"python","sourceType":"notebook"},"papermill":{"default_parameters":{},"duration":167.887423,"end_time":"2026-08-11T20:07:29.578987+00:00","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-08-11T20:04:41.691564+00:00","version":"2.7.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"0af821d64b134940bbaed161fa896caa":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"0f5ccb383f7443aeb4f97fb17fad8253":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"12bd1c6dcae84ddfbc8d323bb0e761a8":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"2559b01750cd4c24aa963ac0374adeb3":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_8456d3130ee14c95963a4c0d55302b5a","max":223,"min":0,"orientation":"horizontal","style":"IPY_MODEL_c77017f511994b7d9498c0c39cd9f3a3","tabbable":null,"tooltip":null,"value":223}},"2771e362da1743b1abb8ceafad4c3e53":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"3136f953745b4a8780af64d4310ac1c5":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"3173cef5754b4184bcd90bf2bb7f5855":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"31892e79776c4f16a40da312aa552f58":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"33f449488ecc486baba59ddff83dc408":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"3439c8f0ab5242199564048e0f4993b4":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_44ce40fc67644f348bf569c66291b0eb","IPY_MODEL_2559b01750cd4c24aa963ac0374adeb3","IPY_MODEL_d5075cbc8a0d47b4825be9c6aee0d04e"],"layout":"IPY_MODEL_31892e79776c4f16a40da312aa552f58","tabbable":null,"tooltip":null}},"36e249c2f01043f4b7c7acb6822f306e":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"44ce40fc67644f348bf569c66291b0eb":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_7799b9adb9e449c1a1c050d5e3d1908f","placeholder":"​","style":"IPY_MODEL_49ef151f7fa44f4fadc57670359aa3b3","tabbable":null,"tooltip":null,"value":"Loading weights: 100%"}},"459503cb760d4a36b74c91edee9456ef":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"45d4ae5783b9440c9a02ce0a43b4ad76":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"49ef151f7fa44f4fadc57670359aa3b3":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"4a607d99c1124a2bbd6ac88d31e9bcbf":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_c9586430915c4967a45e2042ca88dabc","max":223,"min":0,"orientation":"horizontal","style":"IPY_MODEL_45d4ae5783b9440c9a02ce0a43b4ad76","tabbable":null,"tooltip":null,"value":223}},"4c955f0a893a41f7b6b5330ba7a873e3":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_946636428699442d866bc5eb2e75c877","placeholder":"​","style":"IPY_MODEL_33f449488ecc486baba59ddff83dc408","tabbable":null,"tooltip":null,"value":" 223/223 [00:00&lt;00:00, 974.73it/s, Materializing param=layernorm.weight]"}},"4e41b93780de40f390b028cafb90104b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_50c9dcd7f3a244a9b06e7c1fc80c9c87","placeholder":"​","style":"IPY_MODEL_7945865e99e84902a94a73a9e8c8a095","tabbable":null,"tooltip":null,"value":" 223/223 [00:00&lt;00:00, 1029.00it/s, Materializing param=layernorm.weight]"}},"50c9dcd7f3a244a9b06e7c1fc80c9c87":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"516aef62403b42528bf868c873f64fdf":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_3173cef5754b4184bcd90bf2bb7f5855","placeholder":"​","style":"IPY_MODEL_d6fe785cc3694a4eafb500885ed1f095","tabbable":null,"tooltip":null,"value":"Loading weights: 100%"}},"55562e2781a54995981d4f2c1c791320":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_a9112aa0ab55444f832f725b62420cc7","placeholder":"​","style":"IPY_MODEL_fe20f7721a9a4865a07f7eee988a192d","tabbable":null,"tooltip":null,"value":"Loading weights: 100%"}},"55f9e8a6ed134d8baa1ba51793a6aa56":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_cc50da047a1748fc97fe26c220d03b04","placeholder":"​","style":"IPY_MODEL_0af821d64b134940bbaed161fa896caa","tabbable":null,"tooltip":null,"value":"Loading weights: 100%"}},"5a22f2bf9ec34410b83f2f07145fa365":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_c877459d4ca64d7183e8cb90685872e2","max":223,"min":0,"orientation":"horizontal","style":"IPY_MODEL_a421f6ce82c149868a45b2b85f5b4fa3","tabbable":null,"tooltip":null,"value":223}},"5c5189040ebf4ff8a21a0506d41eabfe":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_3136f953745b4a8780af64d4310ac1c5","placeholder":"​","style":"IPY_MODEL_74e51db0de894f098e68d0ddef22dcbb","tabbable":null,"tooltip":null,"value":"Loading weights: 100%"}},"7464dd9c44994060baa709d0e775be7c":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_a41e8ac4713d47e48382d99b0eea1dba","placeholder":"​","style":"IPY_MODEL_a717bbd4d0744ae481eaa7b2c288e6a0","tabbable":null,"tooltip":null,"value":" 223/223 [00:00&lt;00:00, 1043.84it/s, Materializing param=layernorm.weight]"}},"74e51db0de894f098e68d0ddef22dcbb":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"7799b9adb9e449c1a1c050d5e3d1908f":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"7945865e99e84902a94a73a9e8c8a095":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"8456d3130ee14c95963a4c0d55302b5a":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"878b0f7807da4769a866f271ea101d85":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_516aef62403b42528bf868c873f64fdf","IPY_MODEL_5a22f2bf9ec34410b83f2f07145fa365","IPY_MODEL_e3ce2ab94b3d4726afc607cde6826097"],"layout":"IPY_MODEL_459503cb760d4a36b74c91edee9456ef","tabbable":null,"tooltip":null}},"8c4297018eff471483d8b90078a9858b":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8ccbe14ac2ed4aafac63e7fe874c43a7":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_b6c983b1181c46819b3390a185cc633d","max":223,"min":0,"orientation":"horizontal","style":"IPY_MODEL_bcab1f239f4c45f698fde15b7f5485d7","tabbable":null,"tooltip":null,"value":223}},"9092583fd0904303b70d48a990687313":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_5c5189040ebf4ff8a21a0506d41eabfe","IPY_MODEL_9397911ffde347fc875754139a6caeff","IPY_MODEL_7464dd9c44994060baa709d0e775be7c"],"layout":"IPY_MODEL_95998b226c124398a69ee7bed980c051","tabbable":null,"tooltip":null}},"9397911ffde347fc875754139a6caeff":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_8c4297018eff471483d8b90078a9858b","max":223,"min":0,"orientation":"horizontal","style":"IPY_MODEL_f6905e4f1dfe49f6a4f28809cd1ad9c4","tabbable":null,"tooltip":null,"value":223}},"946636428699442d866bc5eb2e75c877":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"95998b226c124398a69ee7bed980c051":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"9802802857fc4517bc165f268951d545":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"a41e8ac4713d47e48382d99b0eea1dba":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"a421f6ce82c149868a45b2b85f5b4fa3":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"a717bbd4d0744ae481eaa7b2c288e6a0":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"a9112aa0ab55444f832f725b62420cc7":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"b6c983b1181c46819b3390a185cc633d":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"bcab1f239f4c45f698fde15b7f5485d7":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"bcbd1e43905041b0b487b83d4c5c4e97":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_55562e2781a54995981d4f2c1c791320","IPY_MODEL_8ccbe14ac2ed4aafac63e7fe874c43a7","IPY_MODEL_4c955f0a893a41f7b6b5330ba7a873e3"],"layout":"IPY_MODEL_36e249c2f01043f4b7c7acb6822f306e","tabbable":null,"tooltip":null}},"c2dc6c5c80e2483a895d7d40e0c01711":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"c77017f511994b7d9498c0c39cd9f3a3":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"c877459d4ca64d7183e8cb90685872e2":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"c9586430915c4967a45e2042ca88dabc":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"cc50da047a1748fc97fe26c220d03b04":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"d5075cbc8a0d47b4825be9c6aee0d04e":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_9802802857fc4517bc165f268951d545","placeholder":"​","style":"IPY_MODEL_2771e362da1743b1abb8ceafad4c3e53","tabbable":null,"tooltip":null,"value":" 223/223 [00:00&lt;00:00, 1033.70it/s, Materializing param=layernorm.weight]"}},"d6fe785cc3694a4eafb500885ed1f095":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"dc0d9fb5781143ad8a37fd20e755b319":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_55f9e8a6ed134d8baa1ba51793a6aa56","IPY_MODEL_4a607d99c1124a2bbd6ac88d31e9bcbf","IPY_MODEL_4e41b93780de40f390b028cafb90104b"],"layout":"IPY_MODEL_c2dc6c5c80e2483a895d7d40e0c01711","tabbable":null,"tooltip":null}},"e3ce2ab94b3d4726afc607cde6826097":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_0f5ccb383f7443aeb4f97fb17fad8253","placeholder":"​","style":"IPY_MODEL_12bd1c6dcae84ddfbc8d323bb0e761a8","tabbable":null,"tooltip":null,"value":" 223/223 [00:00&lt;00:00, 1042.49it/s, Materializing param=layernorm.weight]"}},"f6905e4f1dfe49f6a4f28809cd1ad9c4":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"fe20f7721a9a4865a07f7eee988a192d":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# RSNA Knee — Notebook 03: Hard-Example Fine-Tuning (FT-B)\n\nThis stage continues from the **actual Notebook 02 output**.\n\nNotebook 02 showed an important trade-off:\n\n- Expert-CV BCE improved from about **0.3070 → 0.2717**.\n- Expert-CV macro AUC moved from about **0.9790 → 0.9724**.\n- All 5 conservative FT-A checkpoints passed the BCE promotion gate.\n\nThat means FT-A improved probability calibration, but ranking quality on the tiny 58-study expert set did not improve overall. FT-B therefore changes the objective:\n\n1. Mine hard studies **inside each training fold only**, using that fold's own FT-A parent predictions on its training partition.\n2. Give more sampling probability to confident mistakes, high-loss studies, borderline cases, and rare-positive cases.\n3. Keep the current validation fold completely out of hard-example mining.\n4. Use a smaller update than FT-A: head + last **1 DINO block** with lower learning rates.\n5. Make checkpoint promotion **AUC-first**, with BCE only as a tie-break/safety constraint.\n6. Roll back to the FT-A parent whenever hard-example training fails the gate.\n\nStage 2 OOF predictions are reused only for **diagnostics and reproducibility checks**, not to create the training weights. This avoids subtle cross-fold leakage.\n\n### Submission-file policy\n\nThis notebook does **not** create `submission.csv`.\n\nOnly the final submission notebook will write exactly:\n\n`/kaggle/working/submission.csv`\n\nso there is one unambiguous file to submit after OOF ensemble selection.","metadata":{}},{"cell_type":"markdown","source":"### Fixed version — fingerprint seed contract\n\nThis version fixes the Stage-2 → Stage-3 checkpoint verification bug.\n\n- FT-A checkpoints were fingerprinted with the Notebook 02 stage seed.\n- FT-B uses a different training seed for model diversity.\n- Parent fingerprint verification now reuses the **FT-A fingerprint seed** from the Stage 2 manifest.\n- Every FT-B checkpoint saves its own `fingerprint_seed` for Notebook 04.\n\nThe training seed and the fingerprint-verification seed are therefore no longer accidentally conflated.","metadata":{}},{"cell_type":"code","source":"from __future__ import annotations\n\nimport os\nfor _v in (\"OMP_NUM_THREADS\", \"OPENBLAS_NUM_THREADS\", \"MKL_NUM_THREADS\"):\n    os.environ.setdefault(_v, \"4\")\n\nimport gc\nimport json\nimport math\nimport re\nimport shutil\nimport time\nimport traceback\nfrom concurrent.futures import ThreadPoolExecutor\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom IPython.display import display\nfrom sklearn.metrics import roc_auc_score\n\nT0 = time.time()\nSEED = int(os.environ.get(\"HFT_SEED\", \"31415\"))\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\nif torch.cuda.is_available():\n    torch.cuda.manual_seed_all(SEED)\n\nTARGETS = [\n    \"ACL\", \"MCL\", \"Medial Meniscus\", \"Lateral Meniscus\", \"Medial OA\",\n    \"Lateral OA\", \"PF OA\", \"Effusion\", \"Synovitis\", \"Baker's\",\n    \"Contusion\", \"Fracture\",\n]\n\n# FT-B is intentionally gentler than FT-A.\nHFT_UNFREEZE_LAST = int(os.environ.get(\"HFT_UNFREEZE_LAST\", \"1\"))\nHFT_MAX_EPOCHS = int(os.environ.get(\"HFT_MAX_EPOCHS\", \"3\"))\nHFT_PATIENCE = int(os.environ.get(\"HFT_PATIENCE\", \"2\"))\nHFT_HEAD_LR = float(os.environ.get(\"HFT_HEAD_LR\", \"8e-5\"))\nHFT_BACKBONE_LR = float(os.environ.get(\"HFT_BACKBONE_LR\", \"3e-6\"))\nHFT_WEIGHT_DECAY = float(os.environ.get(\"HFT_WEIGHT_DECAY\", \"0.01\"))\nHFT_BATCH = int(os.environ.get(\"HFT_BATCH\", \"2\"))\nHFT_GRAD_ACCUM = int(os.environ.get(\"HFT_GRAD_ACCUM\", \"2\"))\n\n# Hard mining: at most ~3x sampling weight for the hardest study.\nHFT_HARD_BOOST = float(os.environ.get(\"HFT_HARD_BOOST\", \"2.0\"))\nHFT_DRAW_MULT = float(os.environ.get(\"HFT_DRAW_MULT\", \"1.50\"))\nHFT_HARD_QUANTILE = float(os.environ.get(\"HFT_HARD_QUANTILE\", \"0.70\"))\n\n# AUC-first promotion gate.\nHFT_MIN_AUC_GAIN = float(os.environ.get(\"HFT_MIN_AUC_GAIN\", \"0.001\"))\nHFT_AUC_TIE = float(os.environ.get(\"HFT_AUC_TIE\", \"0.001\"))\nHFT_BCE_TIE_GAIN = float(os.environ.get(\"HFT_BCE_TIE_GAIN\", \"0.003\"))\nHFT_MAX_BCE_WORSEN = float(os.environ.get(\"HFT_MAX_BCE_WORSEN\", \"0.03\"))\n\nEVAL_BATCH = int(os.environ.get(\"HFT_EVAL_BATCH\", \"4\"))\nTIME_BUDGET = float(os.environ.get(\"HFT_TIME_BUDGET_HOURS\", \"8\")) * 3600\n\n# Checkpoint-compatible defaults. Manifest pixel config overwrites these.\nCROP_MM = 130.0\nCACHE_IMG = 336\nIMG = CACHE_IMG\nGROUP = 3\nCACHE_SLICES = 12\nN_GROUP = max(CACHE_SLICES // GROUP, 1)\nSLICE_BAND = (0.20, 0.80)\n\nHDR_THREADS = int(os.environ.get(\"HDR_THREADS\", \"16\"))\nPIX_THREADS = int(os.environ.get(\"PIX_THREADS\", \"12\"))\nORDER_THREADS = int(os.environ.get(\"ORDER_THREADS\", \"32\"))\nORDER_BUDGET_S = 5400\nLAT_MIN_OFFSET_MM = 20.0\nLEGACY_LAT_OFFSET_MM = 5.0\n\nRULES_NATIVE = {\n    \"order\": \"normal\",\n    \"lat\": \"centre\",\n    \"slot_fallback\": False,\n    \"decode_fill\": \"nearest\",\n}\nRULES_LEGACY = {\n    \"order\": \"dominant_axis\",\n    \"lat\": \"corner_x\",\n    \"slot_fallback\": True,\n    \"decode_fill\": \"zero\",\n}\nRULES = dict(RULES_NATIVE)\n\nSLOTS_RECOVERED = [\n    (\"SAG_FLUID_FS\", \"Sagittal\", True, True),\n    (\"COR_FLUID_FS\", \"Coronal\", True, True),\n    (\"AX_FLUID_FS\", \"Axial\", True, True),\n    (\"SAG_FLUID_NOFS\", \"Sagittal\", True, False),\n    (\"COR_T1\", \"Coronal\", False, False),\n    (\"SAG_T1\", \"Sagittal\", False, False),\n]\nSLOTS_PUBLIC = [\n    (\"SAG_FLUID\", \"Sagittal\", None, True),\n    (\"COR_FLUID\", \"Coronal\", None, True),\n    (\"AX_FLUID\", \"Axial\", None, True),\n    (\"SAG_STRUCT\", \"Sagittal\", None, False),\n    (\"COR_STRUCT\", \"Coronal\", None, False),\n    (\"AX_STRUCT\", \"Axial\", None, False),\n]\nSLOTS = SLOTS_RECOVERED\nN_SLOT = len(SLOTS)\n\nPOOL_PARTS = {\"cls_mean\": 2, \"cls_mean_focal\": 3}\nSLOT_PRIOR_TABLE = {\n    \"ACL\": (0, 3, 5), \"MCL\": (1, 4),\n    \"Medial Meniscus\": (0, 1, 3, 4), \"Lateral Meniscus\": (0, 1, 3, 4),\n    \"Medial OA\": (1, 4, 5), \"Lateral OA\": (1, 4, 5),\n    \"PF OA\": (0, 2, 5), \"Effusion\": (0, 2), \"Synovitis\": (0, 2),\n    \"Baker's\": (0,), \"Contusion\": (0, 1, 2), \"Fracture\": (0, 1, 2, 4, 5),\n}\nSLOT_PRIOR_STRENGTH = 0.55\n\nFATSAT_OPTS = {\"FS\", \"FATSAT\", \"FAT_SAT\", \"FSAT\"}\n_SEP = re.compile(r\"[_\\-.]\")\n_FATSAT_RX = re.compile(\n    r\"\\bfs\\b|fatsat|fat sat|\\bstir\\b|\\bspair\\b|\\bspir\\b|\\bwe\\b|\"\n    r\"water excit|\\btirm\\b|\\bsting\\b|\\bfatsup\\b\"\n)\n_T1_RX = re.compile(r\"\\bt1\\b|\\bt1w\\b\")\n_T2_RX = re.compile(r\"\\bt2\\b|\\bt2w\\b\")\n_PD_RX = re.compile(r\"\\bpd\\b|\\bpdw\\b|proton|\\bdp\\b|dens\")\n\nFRONTIER_TARGET_POOL = {\n    \"Fracture\": \"max\",\n    \"Contusion\": \"max\",\n    \"Medial Meniscus\": \"max\",\n    \"Lateral Meniscus\": \"max\",\n    \"ACL\": \"top2\",\n    \"MCL\": \"top2\",\n    \"Baker's\": \"max\",\n}\n\nOUT = Path(\"/kaggle/working/rsna_stage3_ftb\")\nOUT.mkdir(parents=True, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:03.313531Z","iopub.execute_input":"2026-08-11T20:33:03.313826Z","iopub.status.idle":"2026-08-11T20:33:03.332887Z","shell.execute_reply.started":"2026-08-11T20:33:03.3138Z","shell.execute_reply":"2026-08-11T20:33:03.331922Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def log(msg):\n    print(f\"[{time.time() - T0:7.1f}s] {msg}\", flush=True)\n\n\nclass WeightsError(RuntimeError):\n    pass\n\n\ndef find_train_root():\n    explicit = os.environ.get(\"KNEE_INPUT_DIR\", \"\").strip()\n    candidates = [Path(explicit)] if explicit else []\n    candidates += [\n        Path(\"/kaggle/input/competitions/rsna-knee-abnormality-detection\"),\n        Path(\"/kaggle/input/rsna-knee-abnormality-detection\"),\n    ]\n\n    for c in candidates:\n        if (c / \"train.csv\").is_file() and (c / \"train_series\").is_dir():\n            return c\n\n    base = Path(\"/kaggle/input\")\n    if base.is_dir():\n        for root, dirs, files in os.walk(base):\n            dirs[:] = [d for d in dirs if d not in (\"train_series\", \"test_series\")]\n            root = Path(root)\n            if (root / \"train.csv\").is_file() and (root / \"train_series\").is_dir():\n                return root\n\n    raise FileNotFoundError(\"RSNA Knee competition training mount not found\")\n\n\ndef _looks_like_dinov2_config(config_path: Path):\n    try:\n        cfg = json.loads(config_path.read_text())\n    except Exception:\n        return False\n    text = json.dumps(cfg).lower()\n    return (\n        cfg.get(\"model_type\", \"\").lower() == \"dinov2\"\n        or \"dinov2\" in text\n        or any(\"dinov2\" in str(x).lower() for x in cfg.get(\"architectures\", []) or [])\n    )\n\n\ndef find_dinov2(variant=\"small\"):\n    explicit = os.environ.get(\"KNEE_DINOV2_DIR\", \"\").strip()\n    if explicit:\n        p = Path(explicit)\n        if (p / \"config.json\").is_file():\n            return p\n        raise FileNotFoundError(f\"KNEE_DINOV2_DIR={explicit!r} has no config.json\")\n\n    # Prefer the exact official Kaggle DINOv2-small model mount used by the\n    # original inference notebook and Notebook 02. This avoids accidentally\n    # selecting another 384-hidden-size DINO-like config from an auxiliary dataset.\n    preferred = Path(\"/kaggle/input/models/metaresearch/dinov2/pytorch/small/1\")\n    if str(variant).lower() == \"small\" and (preferred / \"config.json\").is_file():\n        return preferred\n\n    base = Path(\"/kaggle/input\")\n    strong, weak = [], []\n\n    if base.is_dir():\n        for root, dirs, files in os.walk(base):\n            dirs[:] = [d for d in dirs if d not in (\"train_series\", \"test_series\")]\n            if \"config.json\" not in files:\n                continue\n            p = Path(root)\n            if _looks_like_dinov2_config(p / \"config.json\"):\n                strong.append(p)\n            elif \"dinov2\" in str(p).lower():\n                weak.append(p)\n\n    hits = strong or weak\n    if not hits:\n        return None\n\n    target_hidden = {\"small\": 384, \"base\": 768, \"large\": 1024, \"giant\": 1536}.get(\n        str(variant).lower()\n    )\n    if target_hidden is not None:\n        for p in hits:\n            try:\n                cfg = json.loads((p / \"config.json\").read_text())\n                if int(cfg.get(\"hidden_size\", -1)) == target_hidden:\n                    return p\n            except Exception:\n                pass\n    return hits[0]\n\n\ndef _is_stage2_dir(p: Path):\n    required = [\n        \"stage2_status.json\",\n        \"manifest.json\",\n        \"expert_cv_predictions.csv\",\n        \"expert_cv_target_auc.csv\",\n        \"expert_ft_fold_map.csv\",\n    ]\n    return all((p / x).is_file() for x in required)\n\n\ndef find_stage2():\n    explicit = os.environ.get(\"STAGE2_DIR\", \"\").strip()\n    if explicit:\n        p = Path(explicit)\n        if _is_stage2_dir(p):\n            return p\n        raise FileNotFoundError(f\"STAGE2_DIR={explicit!r} is not a complete Stage 2 output\")\n\n    base = Path(\"/kaggle/input\")\n    if base.is_dir():\n        for root, dirs, files in os.walk(base):\n            dirs[:] = [d for d in dirs if d not in (\"train_series\", \"test_series\")]\n            p = Path(root)\n            if _is_stage2_dir(p):\n                try:\n                    st = json.loads((p / \"stage2_status.json\").read_text())\n                except Exception:\n                    continue\n                if st.get(\"stage\") == \"FT-A_CONSERVATIVE\":\n                    return p\n\n    raise FileNotFoundError(\n        \"Notebook 02 output not found. Add Notebook 02 output as a Kaggle input \"\n        \"or set STAGE2_DIR.\"\n    )\n\n\nROOT = find_train_root()\nSTAGE2 = find_stage2()\nDINO = find_dinov2(\"small\")\n\nlog(f\"competition root: {ROOT}\")\nlog(f\"stage2 input: {STAGE2}\")\nlog(f\"DINOv2: {DINO}\")\n\nif DINO is None:\n    raise FileNotFoundError(\n        \"DINOv2-small Hugging Face model directory not found. \"\n        \"Attach the same DINO input used by Notebook 02.\"\n    )","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:03.334304Z","iopub.execute_input":"2026-08-11T20:33:03.334754Z","iopub.status.idle":"2026-08-11T20:33:03.371646Z","shell.execute_reply.started":"2026-08-11T20:33:03.334725Z","shell.execute_reply":"2026-08-11T20:33:03.370816Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1. Validate the actual Notebook 02 handoff\n\nFT-B uses the **same 58-study expert CV split** produced by Notebook 02.  \nHardness scores are computed only from Notebook 02's out-of-fold predictions.","metadata":{}},{"cell_type":"code","source":"stage2_status = json.loads((STAGE2 / \"stage2_status.json\").read_text())\nstage2_manifest = json.loads((STAGE2 / \"manifest.json\").read_text())\n\nif not stage2_status.get(\"stage2_ok\"):\n    raise RuntimeError(\"Stage 2 did not finish successfully\")\nif stage2_status.get(\"stage\") != \"FT-A_CONSERVATIVE\":\n    raise RuntimeError(f\"Unexpected Stage 2 type: {stage2_status.get('stage')!r}\")\n\nmembers = stage2_manifest.get(\"members\", [])\nif not members:\n    raise RuntimeError(\"Stage 2 manifest has no members\")\n\nexpert = pd.read_csv(\n    STAGE2 / \"expert_ft_fold_map.csv\",\n    dtype={\"StudyInstanceUID\": str},\n)\noof2 = pd.read_csv(\n    STAGE2 / \"expert_cv_predictions.csv\",\n    dtype={\"StudyInstanceUID\": str},\n)\ntarget_diag2 = pd.read_csv(STAGE2 / \"expert_cv_target_auc.csv\")\n\nif set(expert[\"StudyInstanceUID\"]) != set(oof2[\"StudyInstanceUID\"]):\n    raise RuntimeError(\"Stage 2 expert fold map and OOF predictions contain different study IDs\")\n\nprint(\"Stage 2 members:\", len(members))\nprint(\"Expert studies:\", len(expert))\nprint(\"Stage 2 validation mode:\", stage2_status.get(\"validation_mode\"))\nprint(\"Stage 2 base macro AUC:\", stage2_status.get(\"expert_cv_base_macro_auc\"))\nprint(\"Stage 2 FT-A macro AUC:\", stage2_status.get(\"expert_cv_fta_macro_auc\"))\nprint(\"Stage 2 base BCE:\", stage2_status.get(\"expert_cv_base_bce\"))\nprint(\"Stage 2 FT-A BCE:\", stage2_status.get(\"expert_cv_fta_bce\"))\n\ndisplay(target_diag2)\n\n# Fingerprints are deterministic only when the SAME synthetic-input seed is used.\n# Notebook 02 stored its stage seed in manifest[\"history\"].  FT-B has its own\n# training seed, so using HFT_SEED to verify an FT-A fingerprint is incorrect.\ndef get_stage_seed(manifest, stage_name, default=2026):\n    history = manifest.get(\"history\")\n    if not isinstance(history, list):\n        history = [] if history is None else [history]\n    for item in reversed(history):\n        if isinstance(item, dict) and item.get(\"stage\") == stage_name:\n            try:\n                return int(item.get(\"seed\", default))\n            except Exception:\n                return int(default)\n    return int(default)\n\nSTAGE2_FINGERPRINT_SEED = get_stage_seed(\n    stage2_manifest, \"FT-A_CONSERVATIVE\", default=2026\n)\nlog(f\"FT-A parent fingerprint seed: {STAGE2_FINGERPRINT_SEED}\")\n","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:03.485107Z","iopub.execute_input":"2026-08-11T20:33:03.485783Z","iopub.status.idle":"2026-08-11T20:33:03.515555Z","shell.execute_reply.started":"2026-08-11T20:33:03.485752Z","shell.execute_reply":"2026-08-11T20:33:03.514929Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Load the five FT-A parents and recover their pixel configuration","metadata":{}},{"cell_type":"code","source":"def slot_scheme_for_names(names):\n    names = list(names)\n    rec = [x[0] for x in SLOTS_RECOVERED]\n    pub = [x[0] for x in SLOTS_PUBLIC]\n    if names == rec:\n        return SLOTS_RECOVERED\n    if names == pub:\n        return SLOTS_PUBLIC\n    raise WeightsError(f\"Unknown checkpoint slot scheme: {names}\")\n\n\ndef adopt_config_globals(cfg):\n    global IMG, CACHE_IMG, GROUP, CACHE_SLICES, N_GROUP, CROP_MM\n    global SLICE_BAND, RULES, SLOTS, N_SLOT\n\n    CACHE_IMG = IMG = int(cfg[\"img\"])\n    GROUP = int(cfg[\"group\"])\n    CACHE_SLICES = int(cfg[\"slices\"])\n    N_GROUP = max(CACHE_SLICES // GROUP, 1)\n    CROP_MM = float(cfg[\"crop_mm\"])\n    SLICE_BAND = tuple(float(x) for x in cfg[\"band\"])\n\n    rules = cfg.get(\"rules\") or RULES_NATIVE\n    unknown = {\n        k: v for k, v in rules.items()\n        if k not in RULES_NATIVE or v not in (RULES_NATIVE[k], RULES_LEGACY[k])\n    }\n    if unknown:\n        raise WeightsError(f\"unsupported pixel rules in manifest: {unknown}\")\n    RULES = {**RULES_NATIVE, **rules}\n\n    SLOTS = slot_scheme_for_names(cfg[\"slots\"])\n    N_SLOT = len(SLOTS)\n\n\npixel_groups = {}\nfor m in members:\n    if not (STAGE2 / str(m[\"file\"])).is_file():\n        raise FileNotFoundError(STAGE2 / str(m[\"file\"]))\n    pixel_groups.setdefault(m[\"pixel_group\"], []).append(m)\n\ndisplay(pd.DataFrame([\n    {\n        \"id\": m.get(\"id\"),\n        \"base_fold\": m.get(\"fold\"),\n        \"ft_val_fold\": m.get(\"ft_val_fold\"),\n        \"ft_a_promoted\": m.get(\"ft_promoted\"),\n        \"file\": m.get(\"file\"),\n    }\n    for m in members\n]).sort_values(\"base_fold\"))\n\nprint(\"pixel groups:\", len(pixel_groups))\nfor k, gm in pixel_groups.items():\n    cfg = json.loads(k)\n    print(\n        f\"- {cfg['img']}px, slices={cfg['slices']}, group={cfg['group']}, \"\n        f\"crop={cfg['crop_mm']}mm -> {len(gm)} member(s)\"\n    )","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:03.517059Z","iopub.execute_input":"2026-08-11T20:33:03.517309Z","iopub.status.idle":"2026-08-11T20:33:03.54205Z","shell.execute_reply.started":"2026-08-11T20:33:03.517283Z","shell.execute_reply":"2026-08-11T20:33:03.541426Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Reuse Notebook 02 preprocessing and checkpoint-compatible DINO model\n\nThese cells are carried forward unchanged so a difference in FT-B comes from training strategy,\nnot a silent change in MRI decoding, slot selection, pooling, or model structure.","metadata":{}},{"cell_type":"code","source":"HDR_TAGS = [\"SeriesDescription\", \"SequenceName\", \"ScanOptions\", \"ScanningSequence\",\n            \"RepetitionTime\", \"EchoTime\", \"Laterality\", \"PixelSpacing\", \"Rows\",\n            \"Columns\", \"RescaleSlope\", \"RescaleIntercept\",\n            # Position and orientation are read from the same header probe() already\n            # opens, so they cost nothing, and they are what recovers the side when the\n            # Laterality tag is absent - which it is for half the studies here.\n            \"ImagePositionPatient\", \"ImageOrientationPatient\"]\n\n\ndef _hdr_vec(s, n):\n    \"\"\"Parse a DICOM multi-value string as stored by probe(): floats joined by `|`.\"\"\"\n    if not isinstance(s, str):\n        return None\n    try:\n        v = [float(x) for x in s.split(\"|\")]\n    except ValueError:\n        return None\n    return np.array(v) if len(v) >= n else None\n\n\ndef side_from_geometry(h):\n    \"\"\"Study -> 'L' / 'R' / None, from where the image sits in the patient.\n\n    `Laterality` (0020,0060) is Type 2C and may legitimately be absent; in this corpus it\n    is missing on exactly half the studies, and the vendors it is missing from are whole\n    vendors rather than scattered series. A study with no tag is not a left knee, but the\n    normalisation upstream treats it as one, so half the corpus was never normalised and\n    the five side-defined targets - the two menisci, the two tibiofemoral compartments\n    and the medial collateral ligament - saw that axis reversed on a large minority of it.\n\n    The patient coordinate system fixes this without the tag: +x is the patient's left, so\n    the centre of a right knee sits at negative x. The centre is used rather than\n    `ImagePositionPatient` itself because that is the corner of the image, which is offset\n    by half a field of view - enough to change the sign on a knee near the midline.\n\n    The median over a study's series is what is thresholded, not a single series: probe()\n    reads one arbitrary slice per series, which on a sagittal stack can sit anywhere\n    across the joint. Studies whose centre falls near the midline are left unresolved\n    rather than guessed - measured against the tagged half, the rule is right 97% of the\n    time overall and no better than chance inside 20 mm.\n    \"\"\"\n    cx = {}\n    for r in h.itertuples(index=False):\n        ipp = _hdr_vec(getattr(r, \"ImagePositionPatient\", None), 3)\n        iop = _hdr_vec(getattr(r, \"ImageOrientationPatient\", None), 6)\n        ps = _hdr_vec(getattr(r, \"PixelSpacing\", None), 2)\n        rows, cols = getattr(r, \"Rows\", None), getattr(r, \"Columns\", None)\n        if ipp is None or iop is None or ps is None or not rows or not cols:\n            continue\n        try:\n            c = ipp[:3] + iop[:3] * ps[1] * float(cols) / 2 + iop[3:6] * ps[0] * float(rows) / 2\n        except (TypeError, ValueError):\n            continue\n        cx.setdefault(r.StudyInstanceUID, []).append(float(c[0]))\n    out = {}\n    for st, xs in cx.items():\n        m = float(np.median(xs))\n        out[st] = None if abs(m) < LAT_MIN_OFFSET_MM else (\"R\" if m < 0 else \"L\")\n    return out\n\n\ndef side_from_corner_x(h):\n    \"\"\"The laterality an imported member was fitted under.\n\n    It thresholds the median raw `ImagePositionPatient` x over a study's series. That is\n    the x of the image *corner*, not of its centre, so it differs from the rule above by\n    up to half a field of view - which is enough to reverse the sign on a knee scanned\n    near the midline. The dead zone is 5 mm rather than 20 mm, so it also commits on\n    studies the rule above leaves unresolved.\n\n    Neither difference changes a shape. Each one decides whether a study is mirrored, and\n    a study mirrored one way at training and the other at inference presents the five\n    side-defined targets with their axis reversed.\n    \"\"\"\n    out = {}\n    for st, g in h.groupby(\"StudyInstanceUID\"):\n        xs = []\n        for r in g.itertuples(index=False):\n            ipp = _hdr_vec(getattr(r, \"ImagePositionPatient\", None), 3)\n            if ipp is not None and np.isfinite(ipp).all():\n                xs.append(float(ipp[0]))\n        if not xs:\n            out[st] = None\n            continue\n        x = float(np.median(xs))\n        # DICOM patient coordinates are LPS: +x is the patient's left.\n        out[st] = None if abs(x) < LEGACY_LAT_OFFSET_MM else (\"R\" if x < 0 else \"L\")\n    return out\n\n\ndef lat_of(h, tag=\"\"):\n    \"\"\"Study -> 'L' / 'R' / None: the tag where it exists, geometry where it does not.\n\n    The tag is present on exactly half the studies here and is sometimes an empty\n    string rather than absent, which is not the same as NaN. Treating the other half\n    as left-sided is what `normalise_laterality` did by omission, so the geometry\n    fallback is not a refinement - it is the difference between normalising half the\n    corpus and normalising all of it.\n    \"\"\"\n    geo = side_from_corner_x(h) if RULES[\"lat\"] == \"corner_x\" else side_from_geometry(h)\n    d, n_tag, n_geo, n_none, n_disagree = {}, 0, 0, 0, 0\n    for st, g in h.groupby(\"StudyInstanceUID\"):\n        v = [str(x).strip().upper() for x in g[\"Laterality\"].dropna()]\n        if RULES[\"lat\"] == \"corner_x\" and \"ImageLaterality\" in g.columns:\n            # The legacy rule reads the second tag too, so a study tagged only there is\n            # resolved from the tag rather than from geometry.\n            v += [str(x).strip().upper() for x in g[\"ImageLaterality\"].dropna()]\n        v = [x[0] for x in v if x and x[0] in (\"L\", \"R\")]\n        side = v[0] if v else None\n        if side is not None:\n            n_tag += 1\n            if geo.get(st) is not None and geo[st] != side:\n                n_disagree += 1\n        else:\n            side = geo.get(st)\n            n_geo += side is not None\n            n_none += side is None\n        d[st] = side\n    log(f\"{tag}laterality: {n_tag} from the tag, {n_geo} from geometry, \"\n        f\"{n_none} unresolved; tag and geometry disagree on {n_disagree} \"\n        f\"({n_disagree / max(n_tag, 1):.1%} of the tagged)\")\n    return d\n\n\n\ndef probe(item):\n    split, study, series, path = item\n    row = {\"split\": split, \"StudyInstanceUID\": study, \"SeriesInstanceUID\": series,\n           \"dir\": path}\n    try:\n        files = sorted(e.name for e in os.scandir(path) if e.name.endswith(\".dcm\"))\n        row[\"files\"] = files\n        row[\"n_slices\"] = len(files)\n        if not files:\n            return row\n        ds = pydicom.dcmread(os.path.join(path, files[len(files) // 2]),\n                             stop_before_pixels=True, force=True)\n        for t in HDR_TAGS:\n            v = getattr(ds, t, None)\n            if v is None:\n                row[t] = None\n            elif isinstance(v, (list, tuple)) or type(v).__name__ == \"MultiValue\":\n                row[t] = \"|\".join(str(x) for x in v)\n            else:\n                row[t] = str(v)\n    except Exception as exc:\n        row[\"err\"] = str(exc)[:120]\n    return row\n\n\ndef walk(split):\n    \"\"\"Every series directory of a split, with one header read per series.\n\n    An absent split returns an empty frame *with the columns annotate expects*. Returning\n    a bare DataFrame looks like the same thing and is not: the next call indexes\n    `SeriesDescription` and raises KeyError, so the branch that exists to survive a\n    missing split is what turns it into a crash.\n    \"\"\"\n    base = ROOT / split\n    items = []\n    if not base.is_dir():\n        return pd.DataFrame(columns=[\"split\", \"StudyInstanceUID\", \"SeriesInstanceUID\",\n                                     \"dir\", \"files\", \"n_slices\"] + HDR_TAGS)\n    for study in os.scandir(base):\n        if study.is_dir():\n            for series in os.scandir(study.path):\n                if series.is_dir():\n                    items.append((split, study.name, series.name, series.path))\n    with ThreadPoolExecutor(max_workers=HDR_THREADS) as pool:\n        rows = list(pool.map(probe, items))\n    return pd.DataFrame(rows)\n\n\ndef annotate(df):\n    \"\"\"Recover fat suppression and pulse-sequence weighting from the header.\"\"\"\n    desc = (df[\"SeriesDescription\"].fillna(\"\") + \" \" + df[\"SequenceName\"].fillna(\"\"))\n    desc = desc.str.lower().str.replace(_SEP, \" \", regex=True)\n\n    opts = df[\"ScanOptions\"].fillna(\"\").str.upper().str.split(\"|\")\n    # GE writes SAT_GEMS for spatial saturation, so ScanOptions must be matched as\n    # exact tokens; a substring test on \"SAT\" fires on non-fat-sat series.\n    opts_fs = opts.apply(lambda ts: any(t.strip() in FATSAT_OPTS for t in ts))\n    df[\"fatsat\"] = desc.str.contains(_FATSAT_RX) | opts_fs\n\n    tr = pd.to_numeric(df[\"RepetitionTime\"], errors=\"coerce\")\n    te = pd.to_numeric(df[\"EchoTime\"], errors=\"coerce\")\n    gre = df[\"ScanningSequence\"].fillna(\"\").str.upper().str.contains(\"GR\")\n    t1, t2, pdw = desc.str.contains(_T1_RX), desc.str.contains(_T2_RX), desc.str.contains(_PD_RX)\n\n    df[\"weight\"] = np.where(t1 & ~t2 & ~pdw, \"T1\",\n                     np.where(t2 & ~pdw, \"T2\",\n                       np.where(pdw, \"PD\",\n                         np.where(gre, \"GRE\",\n                           np.where(tr < 800, \"T1\",\n                             np.where(te > 60, \"T2\",\n                               np.where(tr >= 800, \"PD\", \"UNK\")))))))\n    df[\"fluid\"] = np.isin(df[\"weight\"], [\"PD\", \"T2\"])\n    df[\"px\"] = pd.to_numeric(\n        df[\"PixelSpacing\"].fillna(\"\").str.split(\"|\").str[0].replace(\"\", np.nan),\n        errors=\"coerce\")\n    return df","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:03.542997Z","iopub.execute_input":"2026-08-11T20:33:03.543331Z","iopub.status.idle":"2026-08-11T20:33:03.569719Z","shell.execute_reply.started":"2026-08-11T20:33:03.543304Z","shell.execute_reply":"2026-08-11T20:33:03.568913Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def walk_selected(split, study_ids):\n    study_ids = {str(x) for x in study_ids}\n    base = ROOT / split\n    items = []\n\n    if not base.is_dir():\n        raise FileNotFoundError(base)\n\n    for study_id in sorted(study_ids):\n        study_dir = base / study_id\n        if not study_dir.is_dir():\n            continue\n        for series in os.scandir(study_dir):\n            if series.is_dir():\n                items.append((split, study_id, series.name, series.path))\n\n    with ThreadPoolExecutor(max_workers=HDR_THREADS) as pool:\n        rows = list(pool.map(probe, items))\n\n    out = pd.DataFrame(rows)\n    if out.empty:\n        raise RuntimeError(f\"No DICOM series found for {len(study_ids)} selected studies\")\n    return out","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:03.570736Z","iopub.execute_input":"2026-08-11T20:33:03.571316Z","iopub.status.idle":"2026-08-11T20:33:03.586436Z","shell.execute_reply.started":"2026-08-11T20:33:03.571288Z","shell.execute_reply":"2026-08-11T20:33:03.585522Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def _as_bool(v):\n    if pd.isna(v):\n        return None\n    s = str(v).strip().upper()\n    if s in {\"1\", \"TRUE\", \"T\", \"YES\", \"Y\"}:\n        return True\n    if s in {\"0\", \"FALSE\", \"F\", \"NO\", \"N\"}:\n        return False\n    try:\n        f = float(s)\n        return True if f == 1 else False if f == 0 else None\n    except Exception:\n        return None\n\n\ndef audit_official_sequence_metadata(inferred, official):\n    \"\"\"Audit only. Do not change imported checkpoint pixels from this table.\"\"\"\n    need = {\"SeriesInstanceUID\", \"Fluid_Sensitive\", \"Fat_Suppression\"}\n    if inferred.empty or official.empty or not need.issubset(official.columns):\n        return\n    a = inferred[[\"SeriesInstanceUID\", \"fluid\", \"fatsat\"]].copy()\n    b = official[[\"SeriesInstanceUID\", \"Fluid_Sensitive\", \"Fat_Suppression\"]].copy()\n    b[\"official_fluid\"] = b[\"Fluid_Sensitive\"].map(_as_bool)\n    b[\"official_fatsat\"] = b[\"Fat_Suppression\"].map(_as_bool)\n    m = a.merge(b[[\"SeriesInstanceUID\", \"official_fluid\", \"official_fatsat\"]],\n                on=\"SeriesInstanceUID\", how=\"inner\")\n    for inferred_col, official_col, name in [\n        (\"fluid\", \"official_fluid\", \"Fluid_Sensitive\"),\n        (\"fatsat\", \"official_fatsat\", \"Fat_Suppression\"),\n    ]:\n        valid = m[official_col].notna() & m[inferred_col].notna()\n        if valid.any():\n            agree = (m.loc[valid, inferred_col].astype(bool).values ==\n                     m.loc[valid, official_col].astype(bool).values).mean()\n            log(f\"metadata audit {name}: {agree:.1%} agreement on {int(valid.sum())} series\")","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:03.588017Z","iopub.execute_input":"2026-08-11T20:33:03.588462Z","iopub.status.idle":"2026-08-11T20:33:03.599316Z","shell.execute_reply.started":"2026-08-11T20:33:03.58843Z","shell.execute_reply":"2026-08-11T20:33:03.598752Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def pick_slots(series_df, plane_map):\n    \"\"\"One series per slot per study.\n\n    Ties are broken toward the stack with the most slices: a thicker stack samples the\n    joint more densely, and the three-slice sampler below benefits from the margin.\n    \"\"\"\n    series_df = series_df.copy()\n    series_df[\"plane\"] = series_df[\"SeriesInstanceUID\"].map(plane_map)\n    out = {}\n    for study, g in series_df.groupby(\"StudyInstanceUID\"):\n        chosen = {}\n        for name, plane, fluid, fs in SLOTS:\n            sel = (g[\"plane\"] == plane) & (g[\"fatsat\"] == fs)\n            # fluid=None means \"do not condition on weighting\" - the public scheme,\n            # where the single provided flag stands in for both axes at once.\n            if fluid is not None:\n                sel &= (g[\"fluid\"] == fluid)\n            cand = g[sel]\n            # A slot with no series matching its predicate stays empty, and no substitute\n            # is admitted from a neighbouring predicate. Relaxing the weighting to fill a\n            # T1 slot would draw from the pool `SAG_FLUID_NOFS` selects from, since that\n            # pool is what remains once the weighting is dropped: over the training corpus\n            # it would put one series in two slots for 2383 of 4407 studies and leave 56%\n            # of the T1 slot holding PD or T2. The presence mask would then assert a\n            # sequence that was never acquired, and the per-diagnosis softmax of §6 would\n            # divide its attention across two identical slots, giving one acquisition\n            # about twice the weight it carries in a study that holds both. The mask is\n            # there to say a slot is absent, which is what an absent slot is.\n            if len(cand) == 0 and RULES[\"slot_fallback\"] and fluid is False:\n                # The relaxation the paragraph above rejects, reproduced because an\n                # imported member was fitted with its T1 slots filled this way: over half\n                # of that member's training studies had a T1 slot holding a series that\n                # is not T1. Leaving those slots empty would present it with a presence\n                # mask it never saw.\n                cand = g[(g[\"plane\"] == plane) & (~g[\"fatsat\"])]\n            if len(cand):\n                chosen[name] = cand.sort_values(\"n_slices\", ascending=False).iloc[0]\n        out[study] = chosen\n    return out","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:03.600131Z","iopub.execute_input":"2026-08-11T20:33:03.600635Z","iopub.status.idle":"2026-08-11T20:33:03.61623Z","shell.execute_reply.started":"2026-08-11T20:33:03.600603Z","shell.execute_reply":"2026-08-11T20:33:03.615435Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ORDER_TAGS = [(0x0020, 0x0032), (0x0020, 0x0037), (0x0020, 0x0013)]\n\n# Series in which at least one sampled slice would not decode. A list rather than a\n# counter because appending is atomic under the reader threads, and reported rather than\n# swallowed: unreported, a decode failure is indistinguishable from a black knee.\nDECODE_FAILED = []\n\n\ndef cache_tag(rules=None):\n    \"\"\"The name a decoded cache is stored under.\n\n    It has to name everything that decides the pixels, not only their dimensions. Two\n    configurations that agree on resolution, slice count, crop and band but disagree on\n    how a slice is chosen produce different arrays of identical shape - so a tag built\n    from the dimensions alone lets the second attach to the first one's file and train\n    against pixels it never asked for, with nothing anywhere reporting a mismatch.\n\n    A native reading keeps the plain name, so caches decoded before the rules existed\n    stay valid; anything else earns a suffix.\n    \"\"\"\n    r = dict(RULES if rules is None else rules)\n    t = (f\"{CACHE_IMG}px_{CACHE_SLICES}sl_{int(CROP_MM)}mm_\"\n         f\"{SLICE_BAND[0]:.2f}-{SLICE_BAND[1]:.2f}\")\n    if {k: r.get(k, v) for k, v in RULES_NATIVE.items()} != RULES_NATIVE:\n        t += \"_\" + hashlib.md5(json.dumps(r, sort_keys=True).encode()).hexdigest()[:6]\n    return t\n\n\ndef _natural_key(name):\n    return tuple(int(x) if x.isdigit() else x.lower()\n                 for x in re.split(r\"(\\d+)\", str(name)))\n\n\ndef _order_dominant_axis(rec):\n    \"\"\"The slice order an imported member was fitted under.\n\n    It sorts on the raw patient coordinate along whichever axis varies most across the\n    stack, rather than on the projection onto the slice normal. The two differ by a sign,\n    not by a formula: measured over this corpus every sagittal series has a slice normal\n    with n_x in [-1.00, -0.98], so p.n is the negative of the raw x this sorts on and the\n    two stacks come out exactly reversed. Because the band sampler truncates rather than\n    rounds, its nine indices are not symmetric about the middle, so nine slices drawn from\n    a twenty-six slice stack under one order share two with the other.\n\n    Missing geometry falls back to `InstanceNumber` and then to a natural sort of the file\n    name, both at the same 80% threshold the imported pipeline used.\n    \"\"\"\n    files, d = rec[\"files\"], rec[\"dir\"]\n    rows = []\n    for pos, f in enumerate(files):\n        ipp = inst = None\n        try:\n            ds = pydicom.dcmread(os.path.join(d, f), force=True, stop_before_pixels=True,\n                                 specific_tags=[\"ImagePositionPatient\", \"InstanceNumber\"])\n            raw = getattr(ds, \"ImagePositionPatient\", None)\n            if raw is not None and len(raw) >= 3:\n                c = np.asarray(raw[:3], dtype=np.float64)\n                if np.isfinite(c).all():\n                    ipp = c\n            n = getattr(ds, \"InstanceNumber\", None)\n            if n is not None:\n                inst = float(n)\n        except Exception:\n            pass\n        rows.append((f, ipp, inst, pos))\n\n    placed = [r for r in rows if r[1] is not None]\n    need = max(2, int(0.8 * len(rows)))\n    if len(placed) >= need:\n        xyz = np.stack([r[1] for r in placed])\n        axis = int(np.argmax(np.ptp(xyz, axis=0)))\n        spare = float(np.nanmedian(xyz[:, axis]))\n        rows.sort(key=lambda r: (float(r[1][axis]) if r[1] is not None else spare,\n                                 r[2] if r[2] is not None else float(\"inf\"), r[3]))\n    elif sum(r[2] is not None for r in rows) >= need:\n        rows.sort(key=lambda r: (r[2] if r[2] is not None else float(\"inf\"), r[3]))\n    else:\n        rows.sort(key=lambda r: _natural_key(r[0]))\n    return [r[0] for r in rows], True\n\n\ndef order_slices(rec):\n    \"\"\"Return the series' files sorted along the through-plane axis.\n\n    A DICOM file name here is a SOP Instance UID, which is assigned arbitrarily. Sorting\n    by it therefore produces an order uncorrelated with anatomy - measured over one\n    series, Spearman between file-name rank and physical position is 0.009, i.e. none.\n    Anything that assumes the file order means something is then operating on noise: the\n    three channels of a \"2.5D\" input are three unrelated views rather than neighbouring\n    slices, \"the middle of the stack\" is a random subset, and reversing slice order to\n    normalise laterality reverses nothing meaningful.\n\n    The physical order is recoverable exactly. Each slice carries its position in patient\n    coordinates and the in-plane axes; projecting the position onto the slice normal\n    gives a signed through-plane coordinate, monotonic along the stack:\n\n        n = r_x  x  r_y ,      k = p . n\n\n    `InstanceNumber` is the fallback. It usually tracks the projection up to sign, but\n    interleaved and multi-echo acquisitions need not number slices in the order they\n    occupy in space - but the projection is signed in patient\n    coordinates, which is what laterality normalisation needs.\n    \"\"\"\n    if RULES[\"order\"] == \"dominant_axis\":\n        return _order_dominant_axis(rec)\n    files, d = rec[\"files\"], rec[\"dir\"]\n    keyed = []\n    for f in files:\n        k = None\n        try:\n            ds = pydicom.dcmread(os.path.join(d, f), force=True, stop_before_pixels=True,\n                                 specific_tags=ORDER_TAGS)\n            iop = np.asarray(ds.ImageOrientationPatient, dtype=float)\n            ipp = np.asarray(ds.ImagePositionPatient, dtype=float)\n            k = float(np.dot(ipp, np.cross(iop[:3], iop[3:])))\n        except Exception:\n            try:\n                k = float(ds.InstanceNumber)\n            except Exception:\n                k = None\n        keyed.append((k, f))\n    if any(k is None for k, _ in keyed):\n        # A series with no usable geometry keeps its arbitrary order; that is worse than\n        # sorting but better than dropping the series, and it is logged as a count.\n        return files, False\n    return [f for _, f in sorted(keyed, key=lambda t: t[0])], True\n\n\ndef read_slot(rec, n_slice=None, out_size=None):\n    \"\"\"`n_slice` physically spread slices from one series, at `out_size` pixels.\n\n    Returns uint8 [n_slice, out, out] normalised per-series to its 1st-99th\n    percentile. Percentiles rather than min/max because MR intensity has no absolute\n    scale and a single bright vessel would otherwise compress the whole dynamic range.\n\n    Reading is the expensive half of this pipeline, so the caller reads once at the\n    largest configuration it needs and derives the smaller ones from the returned buffer\n    rather than re-reading.\n    \"\"\"\n    n_slice = GROUP if n_slice is None else n_slice\n    out_size = IMG if out_size is None else out_size\n    files, d, px = rec.get(\"ordered\") or rec[\"files\"], rec[\"dir\"], rec[\"px\"]\n    n = len(files)\n    if n == 0:\n        return None\n    # Spread the samples over a central band of the stack: the outermost slices of a knee\n    # series are mostly soft tissue outside the joint. The band is a constant rather than\n    # a literal because how much of the stack is worth reading depends on how many slices\n    # are being taken - at three the middle is all that fits, while at sixteen the ends\n    # are worth having, and a Baker cyst sits at the posteromedial end of a sagittal one.\n    lo, hi = int(SLICE_BAND[0] * (n - 1)), int(SLICE_BAND[1] * (n - 1))\n    idx = np.unique(np.linspace(lo, hi, n_slice).astype(int)) if hi > lo else np.array([n // 2])\n    while len(idx) < n_slice:\n        idx = np.append(idx, idx[-1])\n\n    planes = []\n    for i in idx[:n_slice]:\n        try:\n            ds = pydicom.dcmread(os.path.join(d, files[int(i)]), force=True)\n            a = ds.pixel_array.astype(np.float32)\n            sl = float(getattr(ds, \"RescaleSlope\", 1) or 1)\n            ic = float(getattr(ds, \"RescaleIntercept\", 0) or 0)\n            a = a * sl + ic\n        except Exception:\n            a = None                      # no shape is known here; see below\n        planes.append(a)\n\n    # A slice that would not decode has no shape of its own, and inventing one is how a\n    # single unreadable file erases a whole series: a substitute allocated at the resize\n    # target while the decoded slices are still native makes the shape check below take\n    # the substitute as the authority and zero the good slices with it, leaving a black\n    # slot that the presence mask still reports as acquired.\n    #\n    # A failure is instead filled from the nearest slice that did decode - the same\n    # convention the sampler already uses when the band holds fewer distinct slices than\n    # were asked for - and a series where nothing decodes is reported absent, which the\n    # mask can express, rather than black, which it cannot.\n    got = [k for k, p in enumerate(planes) if p is not None]\n    if RULES[\"decode_fill\"] == \"zero\":\n        # What an imported member was fitted with: a failure becomes a zero plane at the\n        # resize target, which the shape check below then propagates to the whole slot.\n        # It is the behaviour the paragraph above describes and rejects, kept here only\n        # because that member's weights were learned against slots blacked out this way.\n        if not got:\n            DECODE_FAILED.append(rec.get(\"SeriesInstanceUID\", d))\n        planes = [np.zeros((out_size, out_size), np.float32) if p is None else p\n                  for p in planes]\n        got = list(range(len(planes)))\n    if not got:\n        DECODE_FAILED.append(rec.get(\"SeriesInstanceUID\", d))\n        return None\n    if len(got) < len(planes):\n        DECODE_FAILED.append(rec.get(\"SeriesInstanceUID\", d))\n        for k, p in enumerate(planes):\n            if p is None:\n                planes[k] = planes[min(got, key=lambda j: abs(j - k))]\n\n    # Slices of one series can still differ in matrix size - multi-echo and some\n    # reformats do - and those are genuinely not stackable.\n    shp = planes[0].shape\n    planes = [p if p.shape == shp else np.zeros(shp, np.float32) for p in planes]\n    vol = np.stack(planes)\n\n    # constant physical extent, then resize: PixelSpacing varies 3.4x across the corpus\n    if px and np.isfinite(px) and px > 0:\n        want = int(round(CROP_MM / px))\n        h, w = shp\n        if 16 < want < min(h, w):\n            cy, cx = h // 2, w // 2\n            half = want // 2\n            vol = vol[:, max(0, cy - half):cy + half, max(0, cx - half):cx + half]\n\n    lo_v, hi_v = np.percentile(vol, [1, 99])\n    vol = np.clip((vol - lo_v) / max(hi_v - lo_v, 1e-6), 0, 1)\n\n    t = torch.from_numpy(np.ascontiguousarray(vol)).unsqueeze(0)\n    t = F.interpolate(t, size=(out_size, out_size), mode=\"bilinear\", align_corners=False)\n    # uint8, not float32. These buffers queue up between the reader threads and the\n    # encoder, and at this size a float32 slot-series is several megabytes. Intensity is\n    # already normalised into [0, 1] here, so eight bits cost nothing that a bilinear\n    # resize has not already cost, and the queue is a quarter the size.\n    return (t.squeeze(0) * 255).round().clamp(0, 255).to(torch.uint8)","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:03.6576Z","iopub.execute_input":"2026-08-11T20:33:03.658266Z","iopub.status.idle":"2026-08-11T20:33:03.68937Z","shell.execute_reply.started":"2026-08-11T20:33:03.658216Z","shell.execute_reply":"2026-08-11T20:33:03.688411Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def normalise_laterality(img, plane, lat):\n    \"\"\"Map every knee onto a left-knee convention.\n\n    Coronal and axial views mirror under a horizontal flip. Sagittal stacks are not\n    mirror images of each other - the slice order runs medial-to-lateral in opposite\n    directions - so the channel order is reversed instead.\n    \"\"\"\n    if lat != \"R\":\n        return img\n    if plane in (\"Coronal\", \"Axial\"):\n        return torch.flip(img, dims=[-1])\n    return torch.flip(img, dims=[0])","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:03.691306Z","iopub.execute_input":"2026-08-11T20:33:03.691709Z","iopub.status.idle":"2026-08-11T20:33:03.716443Z","shell.execute_reply.started":"2026-08-11T20:33:03.691663Z","shell.execute_reply":"2026-08-11T20:33:03.715434Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Where the geometric slice order may be remembered between runs. Unset on the platform,\n# because each run gets a fresh machine and there is nothing to remember; set off it,\n# where the same corpus is cached again at every resolution and slice count and the order\n# is a function of neither. It is opt-in so that the scored run's behaviour is decided by\n# the code rather than by whether a file happens to be lying about.\nORDER_CACHE = os.environ.get(\"RSNA_ORDER_CACHE\") or None\n\n\ndef build_cache(slot_map, plane_map, lat_map, tag):\n    \"\"\"Decode every (study, slot) once into an in-memory uint8 array.\n\n    Fine-tuning revisits the same pixels every epoch. Reading them from the mount each\n    time would make the epoch count a function of I/O rather than of learning, so they\n    are decoded once and held as bytes: intensity has already been normalised into\n    [0, 1], and eight bits cost nothing a bilinear resize has not already cost.\n\n    CACHE_SLICES positions are kept per slot, which the training loop reads as N_GROUP\n    groups of GROUP consecutive channels.\n    \"\"\"\n    studies = sorted(slot_map)\n    sidx = {s: i for i, s in enumerate(studies)}\n    cache = np.zeros((len(studies), N_SLOT, CACHE_SLICES, IMG, IMG), np.uint8)\n    mask = np.zeros((len(studies), N_SLOT), np.float32)\n    log(f\"{tag}: cache {cache.shape} = {cache.nbytes / 1024 ** 3:.1f} GB\")\n\n    jobs = [(st, k, plane, slot_map[st][name])\n            for st in studies\n            for k, (name, plane, _, _) in enumerate(SLOTS)\n            if name in slot_map[st]]\n    n_job = len(jobs)\n\n    # Ordering first, and as its own pass. It reads one header per slice of every chosen\n    # series - far more file opens than the decode that follows - and on a network mount\n    # that is latency, not work, so it gets its own wider pool.\n    t_ord = time.time()\n    n_slice_total = sum(len(j[3][\"files\"]) for j in jobs)\n    log(f\"{tag}: ordering {len(jobs)} slot-series ({n_slice_total} slice headers)\")\n    ok = done = 0\n    CHUNK_O = 1024\n\n    # A remembered order, when one is offered. The projection depends on the DICOM\n    # geometry alone, so it is the same at every resolution and every slice count, and\n    # it costs one header read per slice - the largest single cost in this pass. An entry\n    # is validated by the number of files present, so a tree that has changed under it is\n    # recomputed rather than trusted: order is derived data, and a stale entry would be\n    # invisible in the way that matters most.\n    seen = {}\n    if ORDER_CACHE and Path(ORDER_CACHE).is_file():\n        try:\n            import json as _json\n            seen = _json.loads(Path(ORDER_CACHE).read_text())\n        except (OSError, ValueError):\n            seen = {}\n        hit = 0\n        for _, _, _, rec in jobs:\n            e = seen.get(rec[\"SeriesInstanceUID\"])\n            if e and len(e[\"files\"]) == len(rec[\"files\"]):\n                rec[\"ordered\"] = e[\"files\"]\n                ok += int(e[\"good\"])\n                hit += 1\n        jobs = [j for j in jobs if \"ordered\" not in j[3]]\n        log(f\"{tag}: {hit} slot-series ordered from {ORDER_CACHE}, {len(jobs)} to read\")\n\n    with ThreadPoolExecutor(max_workers=ORDER_THREADS) as pool:\n        for c0 in range(0, len(jobs), CHUNK_O):\n            block = jobs[c0:c0 + CHUNK_O]\n            for (_, _, _, rec), (files, good) in zip(\n                    block, pool.map(lambda j: order_slices(j[3]), block)):\n                rec[\"ordered\"] = files\n                ok += int(good)\n                done += 1\n                if ORDER_CACHE:\n                    seen[rec[\"SeriesInstanceUID\"]] = {\"files\": files, \"good\": bool(good)}\n            # The ceiling is whichever comes first: the pass's own budget, or the share\n            # of what is left of the run that it may take. The second is what makes the\n            # first safe to set generously - a mount slow enough to matter cannot spend\n            # the training time, because the budget shrinks as the run does.\n            budget = min(ORDER_BUDGET_S, max(60.0, (TIME_BUDGET - (time.time() - T0)) * 0.35))\n            if time.time() - t_ord > budget:\n                log(f\"{tag}: ordering budget spent at {done}/{len(jobs)}; \"\n                    f\"the rest keep file order\")\n                break\n    if ORDER_CACHE and done:\n        import json as _json\n        _t = Path(ORDER_CACHE).with_suffix(\".tmp\")\n        _t.write_text(_json.dumps(seen))\n        _t.replace(Path(ORDER_CACHE))\n    log(f\"{tag}: ordered {ok}/{n_job} by geometry \"\n        f\"({n_job - ok} kept arbitrary) in {time.time() - t_ord:.0f}s\")\n\n    jobs = [(st, k, plane, slot_map[st][name])\n            for st in studies\n            for k, (name, plane, _, _) in enumerate(SLOTS)\n            if name in slot_map[st]]\n    log(f\"{tag}: decoding {len(jobs)} slot-series\")\n    n_failed_before = len(DECODE_FAILED)\n\n    CHUNK = 512\n    done = 0\n    with ThreadPoolExecutor(max_workers=PIX_THREADS) as pool:\n        for c0 in range(0, len(jobs), CHUNK):\n            block = jobs[c0:c0 + CHUNK]\n            for (st, k, plane, _), img in zip(\n                    block, pool.map(lambda j: read_slot(j[3], CACHE_SLICES, IMG), block)):\n                done += 1\n                if img is None:\n                    continue\n                cache[sidx[st], k] = normalise_laterality(img, plane,\n                                                          lat_map.get(st)).numpy()\n                mask[sidx[st], k] = 1.0\n            if done % 4096 < CHUNK:\n                log(f\"  {tag} {done}/{len(jobs)}\")\n            if time.time() - T0 > TIME_BUDGET:\n                log(f\"  {tag}: time budget reached during decode\")\n                break\n    n_failed = len(DECODE_FAILED) - n_failed_before\n    log(f\"{tag}: {int(mask.sum())}/{len(jobs)} slots filled\"\n        + (f\"; {n_failed} series had a slice that would not decode\" if n_failed else \"\"))\n    gc.collect()\n    return studies, cache, mask","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:03.717996Z","iopub.execute_input":"2026-08-11T20:33:03.71882Z","iopub.status.idle":"2026-08-11T20:33:03.743712Z","shell.execute_reply.started":"2026-08-11T20:33:03.718782Z","shell.execute_reply":"2026-08-11T20:33:03.742458Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class SlotHead(nn.Module):\n    \"\"\"Per-diagnosis attention over the slot embeddings of one study.\n\n    Each finding is read on particular sequences - cruciates sagittally, collateral\n    ligaments and the meniscal body coronally, patellar cartilage axially - so pooling\n    the slots identically would dilute the one that carries the evidence with the rest.\n\n    The aggregation is deliberately this simple. With a study-level label there is no\n    signal telling the model which part of a study matters, so extra attention\n    parameters below the slot level would have nothing to learn from and would spend\n    their capacity fitting noise.\n    \"\"\"\n\n    def __init__(self, dim, n_slot, n_out, hidden=256, p=0.2, prior=False):\n        super().__init__()\n        self.proj = nn.Sequential(nn.LayerNorm(dim), nn.Linear(dim, hidden), nn.GELU())\n        self.slot_emb = nn.Parameter(torch.randn(n_slot, hidden) * 0.02)\n        self.query = nn.Parameter(torch.randn(n_out, hidden) * 0.02)\n        self.drop = nn.Dropout(p)\n        self.out = nn.Linear(hidden, n_out)\n        self.hidden = hidden\n        # An imported member carries a fixed per-(diagnosis, slot) tilt on the attention\n        # logits, set from the anatomy table below rather than learned. It is a buffer, so\n        # it travels in the state dict and must exist for that member to load; exp(0.55)\n        # gives a preferred slot about 1.73x the weight of an unpreferred one, which\n        # biases the softmax without ever excluding a slot.\n        p_ = torch.zeros(n_out, n_slot)\n        if prior and n_slot == len(SLOTS) and n_out == len(TARGETS):\n            for t, slots in SLOT_PRIOR_TABLE.items():\n                if t in TARGETS:\n                    p_[TARGETS.index(t), list(slots)] = SLOT_PRIOR_STRENGTH\n        self.prior = prior\n        if prior:\n            self.register_buffer(\"slot_prior\", p_)\n\n    def forward(self, x, mask):\n        h = self.proj(x) + self.slot_emb\n        att = torch.einsum(\"bsh,oh->bos\", h, self.query) / self.hidden ** 0.5\n        if self.prior:\n            att = att + self.slot_prior.unsqueeze(0)\n        att = att.masked_fill(mask.unsqueeze(1) < 0.5, -1e4).softmax(-1)\n        ctx = self.drop(torch.einsum(\"bos,bsh->boh\", att, h))\n        return (ctx * self.out.weight.unsqueeze(0)).sum(-1) + self.out.bias","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:03.745043Z","iopub.execute_input":"2026-08-11T20:33:03.745447Z","iopub.status.idle":"2026-08-11T20:33:03.759854Z","shell.execute_reply.started":"2026-08-11T20:33:03.745405Z","shell.execute_reply":"2026-08-11T20:33:03.759086Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Model(nn.Module):\n    \"\"\"Encoder plus head, trained end to end.\n\n    A study arrives as a bag of slot images. The bag is flattened for the encoder and\n    folded back before the head, so the encoder never sees the study structure and the\n    head never sees pixels.\n    \"\"\"\n\n    def __init__(self, backbone, dim, pool=\"cls_mean\", prior=False):\n        super().__init__()\n        self.backbone = backbone\n        self.pool = pool\n        self.head = SlotHead(dim * POOL_PARTS[pool], N_SLOT, len(TARGETS), prior=prior)\n        self.register_buffer(\"mean\", torch.tensor([0.485, 0.456, 0.406]).view(1, 3, 1, 1))\n        self.register_buffer(\"std\", torch.tensor([0.229, 0.224, 0.225]).view(1, 3, 1, 1))\n\n    def forward(self, imgs, mask, img_size=None):\n        B, S = imgs.shape[:2]\n        x = imgs.reshape(B * S, *imgs.shape[2:]).float().div_(255.0)\n        if img_size is not None and img_size != x.shape[-1]:\n            # The cache is held at the highest resolution any configuration needs; the\n            # rest downsample from it, so every configuration sees the same pixels\n            # through a different sampling grid rather than a different crop.\n            x = F.interpolate(x, size=(img_size, img_size), mode=\"bilinear\",\n                              align_corners=False)\n        x = (x - self.mean) / self.std\n        out = self.backbone(pixel_values=x).last_hidden_state\n        patch = out[:, 1:]\n        parts = [out[:, 0], patch.mean(1)]\n        if self.pool == \"cls_mean_focal\":\n            # The upper tail of each channel over the patch grid, taken per channel\n            # rather than by selecting whole patches: a finding occupies a small part of\n            # the field, so a plain mean over 256 patches dilutes it by two orders of\n            # magnitude, and this keeps the top eighth of each channel's responses.\n            k = max(1, patch.shape[1] // 8)\n            parts.append(patch.topk(k, dim=1).values.mean(1))\n        feat = torch.cat(parts, dim=1).reshape(B, S, -1)\n        return self.head(feat, mask)","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:03.762623Z","iopub.execute_input":"2026-08-11T20:33:03.762881Z","iopub.status.idle":"2026-08-11T20:33:03.775843Z","shell.execute_reply.started":"2026-08-11T20:33:03.76285Z","shell.execute_reply":"2026-08-11T20:33:03.774783Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_model(unfreeze_last, source=None, variant=\"small\", pool=\"cls_mean\", prior=False):\n    from transformers import AutoModel\n\n    p = Path(source) if source is not None else DINO\n    if p is None:\n        raise FileNotFoundError(\"DINOv2 weights not attached\")\n\n    bb = AutoModel.from_pretrained(str(p))\n    n_layer = len(bb.encoder.layer)\n\n    for prm in bb.parameters():\n        prm.requires_grad = False\n    for blk in bb.encoder.layer[max(0, n_layer - int(unfreeze_last)):]:\n        for prm in blk.parameters():\n            prm.requires_grad = True\n    for prm in bb.layernorm.parameters():\n        prm.requires_grad = True\n\n    dim = bb.config.hidden_size\n    trainable = sum(p.numel() for p in bb.parameters() if p.requires_grad)\n    log(\n        f\"backbone: {n_layer} blocks, last {unfreeze_last} trainable \"\n        f\"({trainable / 1e6:.1f}M backbone params)\"\n    )\n    return Model(bb, dim, pool=pool, prior=prior)","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:03.777394Z","iopub.execute_input":"2026-08-11T20:33:03.77791Z","iopub.status.idle":"2026-08-11T20:33:03.792419Z","shell.execute_reply.started":"2026-08-11T20:33:03.777871Z","shell.execute_reply":"2026-08-11T20:33:03.791694Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"FINGERPRINT_TOL = 2e-3\n\ndef fingerprint(model, dev, img_size, n_slot=None, group=None, seed=None):\n    n_slot = N_SLOT if n_slot is None else n_slot\n    group = GROUP if group is None else group\n    seed = SEED if seed is None else seed\n\n    g = torch.Generator().manual_seed(seed)\n    imgs = torch.randint(\n        0, 256, (2, n_slot, group, img_size, img_size),\n        generator=g, dtype=torch.uint8\n    ).to(dev)\n    mask = torch.ones(2, n_slot, device=dev)\n    mask[1, -1] = 0.0\n\n    was_training = model.training\n    model.eval()\n    with torch.no_grad():\n        out = model(imgs, mask, img_size).float().cpu().numpy()\n    if was_training:\n        model.train()\n    return out\n\n\ndef check_fingerprint(\n    model,\n    dev,\n    img_size,\n    expected,\n    tol=FINGERPRINT_TOL,\n    tag=\"\",\n    seed=None,\n):\n    # IMPORTANT: a stored fingerprint must be reproduced with the same\n    # deterministic synthetic-input seed that was used when it was saved.\n    got = fingerprint(model, dev, img_size, seed=seed)\n    exp = np.asarray(expected, np.float32)\n    if got.shape != exp.shape:\n        raise WeightsError(f\"{tag}fingerprint shape {got.shape} != stored {exp.shape}\")\n    d = float(np.abs(got - exp).max())\n    if d > tol:\n        raise WeightsError(\n            f\"{tag}fingerprint differs by {d:.4g} > {tol:g}; \"\n            \"architecture/preprocessing contract moved\"\n        )\n    log(f\"{tag}fingerprint matches within {d:.2g} (seed={seed})\")\n    return d\n\n\ndef window_starts(n_slice, group):\n    if n_slice >= group:\n        return list(range(n_slice - group + 1))\n    return [0]\n\n\ndef _apply_frontier_pool(probs):\n    v = probs.mean(dim=0)\n    target_idx = {t: j for j, t in enumerate(TARGETS)}\n    for target, mode in FRONTIER_TARGET_POOL.items():\n        j = target_idx[target]\n        x = probs[:, :, j]\n        if mode == \"max\":\n            v[:, j] = x.max(dim=0).values\n        elif mode.startswith(\"top\"):\n            k = min(int(mode[3:]), x.shape[0])\n            v[:, j] = x.topk(k, dim=0).values.mean(dim=0)\n        elif mode == \"mean\":\n            v[:, j] = x.mean(dim=0)\n        else:\n            raise ValueError(mode)\n    return v\n\n\n@torch.no_grad()\ndef predict_member_frontier(model, cache, mask, idx, dev, img_size):\n    starts = window_starts(cache.shape[2], GROUP)\n    model.eval()\n    pred = np.empty((len(idx), len(TARGETS)), np.float32)\n\n    for b0 in range(0, len(idx), EVAL_BATCH):\n        sel = np.asarray(idx[b0:b0 + EVAL_BATCH], dtype=int)\n        b1 = b0 + len(sel)\n        m = torch.from_numpy(mask[sel]).to(dev)\n        win = []\n\n        for st in starts:\n            rows = torch.from_numpy(\n                np.ascontiguousarray(cache[sel, :, st:st + GROUP])\n            ).to(dev)\n            with torch.autocast(device_type=dev.type, enabled=dev.type == \"cuda\"):\n                logits = model(rows, m, img_size).float()\n            win.append(torch.sigmoid(logits))\n\n        probs = torch.stack(win, dim=0)\n        pred[b0:b1] = _apply_frontier_pool(probs).cpu().numpy()\n\n    return pred","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:03.793642Z","iopub.execute_input":"2026-08-11T20:33:03.79391Z","iopub.status.idle":"2026-08-11T20:33:03.811543Z","shell.execute_reply.started":"2026-08-11T20:33:03.793883Z","shell.execute_reply":"2026-08-11T20:33:03.810804Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Decode the same 58 expert studies once per pixel group","metadata":{}},{"cell_type":"code","source":"train_series = pd.read_csv(\n    ROOT / \"train_series.csv\",\n    dtype={\"StudyInstanceUID\": str, \"SeriesInstanceUID\": str},\n)\nplane_map = dict(zip(\n    train_series[\"SeriesInstanceUID\"].astype(str),\n    train_series[\"Anatomical_Plane\"],\n))\n\nexpert_ids = expert[\"StudyInstanceUID\"].astype(str).tolist()\nhtr = annotate(walk_selected(\"train_series\", expert_ids))\nlog(\n    f\"expert header pass: {len(htr)} series across \"\n    f\"{htr['StudyInstanceUID'].nunique()} studies\"\n)\n\ngroup_cache = {}\n\nfor gi, (pixel_key, gm) in enumerate(pixel_groups.items(), 1):\n    cfg = json.loads(pixel_key)\n    adopt_config_globals(cfg)\n\n    log(\n        f\"decode group {gi}/{len(pixel_groups)}: \"\n        f\"{IMG}px, {CACHE_SLICES} slices, group={GROUP}, crop={CROP_MM}mm\"\n    )\n\n    slot_map = pick_slots(htr, plane_map)\n    studies, cache, slot_mask = build_cache(\n        slot_map,\n        plane_map,\n        lat_of(htr, f\"expert g{gi} \"),\n        f\"expert g{gi}\",\n    )\n\n    group_cache[pixel_key] = {\n        \"studies\": studies,\n        \"cache\": cache,\n        \"slot_mask\": slot_mask,\n        \"cfg\": cfg,\n    }\n\n    log(\n        f\"g{gi}: cache {cache.shape}, slot coverage \"\n        f\"{slot_mask.sum():.0f}/{slot_mask.size}\"\n    )\n\nprint(\"cached pixel groups:\", len(group_cache))","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:03.812405Z","iopub.execute_input":"2026-08-11T20:33:03.812698Z","iopub.status.idle":"2026-08-11T20:33:41.849297Z","shell.execute_reply.started":"2026-08-11T20:33:03.812673Z","shell.execute_reply":"2026-08-11T20:33:41.848646Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Prepare Stage 2 OOF diagnostics — not training weights\n\nThe Stage 2 OOF predictions remain useful for:\n\n- reproducing the FT-A parent on each validation fold;\n- comparing BASE vs FT-A vs FT-B at the end.\n\n**They are not used to weight FT-B training samples.**\n\nHard-example weights are generated fold-locally later, after each FT-A parent predicts only its own training partition.","metadata":{}},{"cell_type":"code","source":"# Align Stage 2 OOF rows to the expert fold-map order.\nexpert = expert.set_index(\"StudyInstanceUID\").loc[\n    oof2[\"StudyInstanceUID\"].astype(str)\n].reset_index()\noof2 = oof2.set_index(\"StudyInstanceUID\").loc[\n    expert[\"StudyInstanceUID\"].astype(str)\n].reset_index()\n\nbase_oof_stage2 = np.column_stack([\n    oof2[f\"base__{t}\"].to_numpy(float) for t in TARGETS\n])\nfta_oof_stage2 = np.column_stack([\n    oof2[f\"fta__{t}\"].to_numpy(float) for t in TARGETS\n])\n\nif not np.isfinite(base_oof_stage2).all() or not np.isfinite(fta_oof_stage2).all():\n    raise RuntimeError(\"Stage 2 OOF file contains missing/non-finite predictions\")\n\nstage2_target_diag = target_diag2.set_index(\"target\").reindex(TARGETS).reset_index()\nstage2_target_diag.to_csv(OUT / \"stage2_target_diagnostic.csv\", index=False)\ndisplay(stage2_target_diag)\n\nprint(\n    \"Hard-example training weights will be mined fold-locally from each FT-A \"\n    \"parent's predictions on its TRAINING partition only.\"\n)","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:41.850402Z","iopub.execute_input":"2026-08-11T20:33:41.850787Z","iopub.status.idle":"2026-08-11T20:33:41.875748Z","shell.execute_reply.started":"2026-08-11T20:33:41.85076Z","shell.execute_reply":"2026-08-11T20:33:41.87502Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. FT-B training utilities: fold-local hard mining + AUC-first rollback gate","metadata":{}},{"cell_type":"code","source":"def align_training_arrays(studies):\n    e = expert.set_index(\"StudyInstanceUID\").loc[list(studies)]\n    y = e[TARGETS].to_numpy(np.float32)\n    label_mask = np.isfinite(y).astype(np.float32)\n    y = np.nan_to_num(y, nan=0.0)\n    folds = e[\"ft_fold\"].to_numpy(np.int64)\n    return y, label_mask, folds\n\n\ndef rank01(x):\n    x = np.asarray(x, dtype=np.float64)\n    return pd.Series(x).rank(method=\"average\", pct=True).to_numpy(np.float64)\n\n\ndef mine_fold_local_hardness(pred, y, label_mask, train_idx):\n    # Build training weights only from current-parent predictions on train_idx.\n    # No current validation labels or predictions enter this calculation.\n    train_idx = np.asarray(train_idx, dtype=int)\n    p = np.clip(pred.astype(np.float64), 1e-5, 1 - 1e-5)\n    yy = y[train_idx].astype(np.float64)\n    mm = label_mask[train_idx].astype(np.float64)\n\n    ce = -(yy * np.log(p) + (1 - yy) * np.log(1 - p))\n    ce = (ce * mm).sum(axis=1) / np.maximum(mm.sum(axis=1), 1.0)\n\n    error_prob = yy * (1 - p) + (1 - yy) * p\n    conf_wrong = np.max(\n        np.where((error_prob > 0.5) & (mm > 0.5), error_prob, 0.0),\n        axis=1,\n    )\n\n    uncertainty = 1.0 - 2.0 * np.abs(p - 0.5)\n    uncertainty = (uncertainty * mm).sum(axis=1) / np.maximum(mm.sum(axis=1), 1.0)\n\n    # Minority-positive bonus computed from this TRAINING partition only.\n    pos_count = np.maximum((yy * mm).sum(axis=0), 1.0)\n    rare_target = 1.0 / np.sqrt(pos_count)\n    minority_bonus = (\n        (yy * mm * rare_target[None, :]).sum(axis=1)\n        / max(rare_target.sum(), 1e-8)\n    )\n\n    raw = (\n        0.60 * rank01(ce)\n        + 0.20 * rank01(conf_wrong)\n        + 0.15 * rank01(uncertainty)\n        + 0.05 * rank01(minority_bonus)\n    )\n    hardness = rank01(raw)\n    sample_weight = 1.0 + HFT_HARD_BOOST * hardness\n    hard_cut = float(np.quantile(hardness, HFT_HARD_QUANTILE))\n    is_hard = hardness >= hard_cut\n\n    top_target_idx = np.argmax(error_prob * mm, axis=1)\n    top_target = [TARGETS[int(j)] for j in top_target_idx]\n    top_target_error = error_prob[np.arange(len(train_idx)), top_target_idx]\n\n    return {\n        \"train_idx\": train_idx,\n        \"ce\": ce,\n        \"conf_wrong\": conf_wrong,\n        \"uncertainty\": uncertainty,\n        \"minority_bonus\": minority_bonus,\n        \"hardness\": hardness,\n        \"sample_weight\": sample_weight.astype(np.float32),\n        \"is_hard\": is_hard,\n        \"hardest_target\": top_target,\n        \"hardest_target_error_prob\": top_target_error,\n    }\n\n\ndef make_pos_weight(y, label_mask, idx):\n    yy = y[idx]\n    mm = label_mask[idx]\n    pos = (yy * mm).sum(axis=0)\n    known = mm.sum(axis=0)\n    neg = known - pos\n    ratio = np.sqrt((neg + 0.5) / (pos + 0.5))\n    ratio = np.clip(ratio, 1.0, 3.0)\n    return torch.tensor(ratio, dtype=torch.float32)\n\n\ndef masked_bce_logits(logits, y, mask, pos_weight):\n    per = F.binary_cross_entropy_with_logits(\n        logits, y, reduction=\"none\", pos_weight=pos_weight\n    )\n    return (per * mask).sum() / mask.sum().clamp_min(1.0)\n\n\ndef prob_bce(pred, y, mask):\n    p = np.clip(np.asarray(pred, np.float64), 1e-6, 1 - 1e-6)\n    y = np.asarray(y, np.float64)\n    m = np.asarray(mask, np.float64)\n    per = -(y * np.log(p) + (1 - y) * np.log(1 - p))\n    return float((per * m).sum() / max(m.sum(), 1.0))\n\n\ndef macro_auc(pred, y, mask):\n    vals = []\n    per_target = {}\n    for j, t in enumerate(TARGETS):\n        valid = mask[:, j] > 0.5\n        yy = y[valid, j]\n        pp = pred[valid, j]\n        if len(np.unique(yy)) < 2:\n            per_target[t] = np.nan\n            continue\n        a = float(roc_auc_score(yy, pp))\n        per_target[t] = a\n        vals.append(a)\n    return (float(np.mean(vals)) if vals else np.nan), per_target\n\n\ndef optimizer_for(model):\n    head = [p for p in model.head.parameters() if p.requires_grad]\n    backbone = [p for p in model.backbone.parameters() if p.requires_grad]\n    return torch.optim.AdamW(\n        [\n            {\"params\": head, \"lr\": HFT_HEAD_LR},\n            {\"params\": backbone, \"lr\": HFT_BACKBONE_LR},\n        ],\n        weight_decay=HFT_WEIGHT_DECAY,\n    )\n\n\ndef cpu_state_dict(model):\n    return {k: v.detach().cpu().clone() for k, v in model.state_dict().items()}\n\n\ndef better_candidate(cand_auc, cand_bce, best_auc, best_bce, parent_bce):\n    if not np.isfinite(cand_auc) or not np.isfinite(best_auc):\n        return cand_bce < best_bce - HFT_BCE_TIE_GAIN\n\n    if (\n        cand_auc > best_auc + HFT_MIN_AUC_GAIN\n        and cand_bce <= parent_bce + HFT_MAX_BCE_WORSEN\n    ):\n        return True\n\n    if (\n        cand_auc >= best_auc - HFT_AUC_TIE\n        and cand_bce < best_bce - HFT_BCE_TIE_GAIN\n        and cand_bce <= parent_bce + HFT_MAX_BCE_WORSEN\n    ):\n        return True\n\n    return False\n\n\ndef train_one_hard_epoch(\n    model,\n    cache,\n    slot_mask,\n    y,\n    label_mask,\n    train_idx,\n    sample_weight,\n    dev,\n    optimizer,\n    scaler,\n    pos_weight,\n    epoch_seed,\n):\n    model.train()\n    rng = np.random.default_rng(epoch_seed)\n\n    train_idx = np.asarray(train_idx, dtype=int)\n    probs = np.asarray(sample_weight, dtype=np.float64)\n    probs = probs / probs.sum()\n\n    draw_n = max(\n        len(train_idx),\n        int(math.ceil(len(train_idx) * HFT_DRAW_MULT)),\n    )\n    order = rng.choice(train_idx, size=draw_n, replace=True, p=probs)\n\n    optimizer.zero_grad(set_to_none=True)\n    losses = []\n    n_steps = math.ceil(len(order) / HFT_BATCH)\n\n    for step, b0 in enumerate(range(0, len(order), HFT_BATCH), 1):\n        sel = order[b0:b0 + HFT_BATCH]\n        starts = rng.integers(\n            0,\n            max(cache.shape[2] - GROUP + 1, 1),\n            size=len(sel),\n        )\n\n        x_np = np.stack(\n            [cache[i, :, int(st):int(st) + GROUP] for i, st in zip(sel, starts)],\n            axis=0,\n        )\n        x = torch.from_numpy(np.ascontiguousarray(x_np)).to(dev)\n        sm = torch.from_numpy(slot_mask[sel]).to(dev)\n        yy = torch.from_numpy(y[sel]).to(dev)\n        lm = torch.from_numpy(label_mask[sel]).to(dev)\n\n        with torch.autocast(device_type=dev.type, enabled=dev.type == \"cuda\"):\n            logits = model(x, sm, IMG)\n            loss = masked_bce_logits(\n                logits,\n                yy,\n                lm,\n                pos_weight.to(dev),\n            )\n            loss = loss / HFT_GRAD_ACCUM\n\n        scaler.scale(loss).backward()\n        losses.append(float(loss.detach().cpu()) * HFT_GRAD_ACCUM)\n\n        if step % HFT_GRAD_ACCUM == 0 or step == n_steps:\n            scaler.unscale_(optimizer)\n            torch.nn.utils.clip_grad_norm_(\n                [p for p in model.parameters() if p.requires_grad],\n                max_norm=1.0,\n            )\n            scaler.step(optimizer)\n            scaler.update()\n            optimizer.zero_grad(set_to_none=True)\n\n    return float(np.mean(losses)) if losses else np.nan","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:41.876832Z","iopub.execute_input":"2026-08-11T20:33:41.87726Z","iopub.status.idle":"2026-08-11T20:33:41.905378Z","shell.execute_reply.started":"2026-08-11T20:33:41.877232Z","shell.execute_reply":"2026-08-11T20:33:41.904662Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. Hard-example fine-tuning loop\n\nEach member keeps the same expert validation fold used in Notebook 02.\n\nThe starting point is its FT-A checkpoint.  \nIf FT-B does not clear the AUC-first gate, the output checkpoint is an exact rollback copy of FT-A.","metadata":{}},{"cell_type":"code","source":"dev = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nlog(f\"device: {dev}\")\n\nmetrics_rows = []\nnew_members = []\nhard_rows = []\n\nuid_order = expert[\"StudyInstanceUID\"].astype(str).tolist()\nuid_to_global = {u: i for i, u in enumerate(uid_order)}\noof_ftb_sum = np.zeros((len(expert), len(TARGETS)), np.float64)\noof_count = np.zeros(len(expert), np.int64)\n\noof2_idx = oof2.set_index(\"StudyInstanceUID\")\n\nfor mi, m in enumerate(members, 1):\n    if time.time() - T0 > TIME_BUDGET:\n        log(\"time budget reached before all Stage 2 parents were processed\")\n        break\n\n    pixel_key = m[\"pixel_group\"]\n    pack = group_cache[pixel_key]\n    pixel_cfg = pack[\"cfg\"]\n    member_cfg = m.get(\"config\") or {}\n    adopt_config_globals(pixel_cfg)\n\n    studies = pack[\"studies\"]\n    cache = pack[\"cache\"]\n    slot_mask = pack[\"slot_mask\"]\n    y, label_mask, ft_folds = align_training_arrays(studies)\n\n    val_fold = int(m.get(\"ft_val_fold\", m.get(\"fold\")))\n    train_idx = np.where(ft_folds != val_fold)[0]\n    val_idx = np.where(ft_folds == val_fold)[0]\n\n    log(\n        f\"[{mi}/{len(members)}] {m['id']} -> FT-B val-fold={val_fold}; \"\n        f\"train={len(train_idx)}, val={len(val_idx)}\"\n    )\n\n    ck_path = STAGE2 / str(m[\"file\"])\n    ck = torch.load(ck_path, map_location=\"cpu\", weights_only=False)\n\n    model = build_model(\n        HFT_UNFREEZE_LAST,\n        source=DINO,\n        variant=member_cfg.get(\"variant\", \"small\"),\n        pool=member_cfg.get(\"pool\", \"cls_mean\"),\n        prior=bool(member_cfg.get(\"prior\", False)),\n    ).to(dev)\n    model.load_state_dict(ck[\"model\"], strict=True)\n\n    if \"fingerprint\" in ck:\n        parent_fp_seed = int(\n            ck.get(\"fingerprint_seed\", STAGE2_FINGERPRINT_SEED)\n        )\n        check_fingerprint(\n            model,\n            dev,\n            IMG,\n            ck[\"fingerprint\"],\n            tag=f\"{m['id']}: \",\n            seed=parent_fp_seed,\n        )\n\n    # Validation prediction: evaluation only.\n    parent_pred = predict_member_frontier(\n        model, cache, slot_mask, val_idx, dev, IMG\n    )\n    parent_bce = prob_bce(parent_pred, y[val_idx], label_mask[val_idx])\n    parent_auc, _ = macro_auc(parent_pred, y[val_idx], label_mask[val_idx])\n\n    # Reproduce Stage 2 validation predictions before continuing.\n    val_uids = [studies[int(i)] for i in val_idx]\n    expected = np.column_stack([\n        oof2_idx.loc[val_uids, f\"fta__{t}\"].to_numpy(float)\n        for t in TARGETS\n    ])\n    parent_oof_maxdiff = float(np.max(np.abs(parent_pred - expected)))\n    log(f\"  parent-vs-Stage2 OOF max |diff| = {parent_oof_maxdiff:.3g}\")\n    if parent_oof_maxdiff > 5e-3:\n        raise WeightsError(\n            f\"{m['id']}: parent prediction mismatch {parent_oof_maxdiff:.4g}; \"\n            \"preprocessing/model contract is not reproducing Stage 2\"\n        )\n\n    # Fold-local hard mining: TRAINING partition only.\n    train_parent_pred = predict_member_frontier(\n        model, cache, slot_mask, train_idx, dev, IMG\n    )\n    hard = mine_fold_local_hardness(\n        train_parent_pred,\n        y,\n        label_mask,\n        train_idx,\n    )\n    study_weight = hard[\"sample_weight\"]\n\n    log(\n        f\"  fold-local hard mining: hard={int(hard['is_hard'].sum())}/{len(train_idx)}, \"\n        f\"sample-weight mean={study_weight.mean():.3f}, max={study_weight.max():.3f}\"\n    )\n\n    for k, cache_idx in enumerate(train_idx):\n        hard_rows.append({\n            \"parent_id\": m[\"id\"],\n            \"ft_val_fold\": val_fold,\n            \"StudyInstanceUID\": studies[int(cache_idx)],\n            \"parent_train_logloss\": float(hard[\"ce\"][k]),\n            \"confident_wrong\": float(hard[\"conf_wrong\"][k]),\n            \"uncertainty\": float(hard[\"uncertainty\"][k]),\n            \"minority_positive_bonus\": float(hard[\"minority_bonus\"][k]),\n            \"hardness\": float(hard[\"hardness\"][k]),\n            \"sample_weight\": float(hard[\"sample_weight\"][k]),\n            \"is_hard\": bool(hard[\"is_hard\"][k]),\n            \"hardest_target\": hard[\"hardest_target\"][k],\n            \"hardest_target_error_prob\": float(hard[\"hardest_target_error_prob\"][k]),\n        })\n\n    best_bce = parent_bce\n    best_auc = parent_auc\n    best_epoch = 0\n    best_state = None\n    best_pred = parent_pred.copy()\n\n    pos_weight = make_pos_weight(y, label_mask, train_idx)\n    optimizer = optimizer_for(model)\n\n    def lr_factor(ep):\n        if HFT_MAX_EPOCHS <= 1:\n            return 1.0\n        x = min(max(ep, 0), HFT_MAX_EPOCHS)\n        return max(0.15, 0.5 * (1.0 + math.cos(math.pi * x / HFT_MAX_EPOCHS)))\n\n    scheduler = torch.optim.lr_scheduler.LambdaLR(\n        optimizer, lr_lambda=[lr_factor, lr_factor]\n    )\n    scaler = torch.amp.GradScaler(\"cuda\", enabled=dev.type == \"cuda\")\n\n    stale = 0\n    history = []\n\n    for epoch in range(1, HFT_MAX_EPOCHS + 1):\n        train_loss = train_one_hard_epoch(\n            model,\n            cache,\n            slot_mask,\n            y,\n            label_mask,\n            train_idx,\n            study_weight,\n            dev,\n            optimizer,\n            scaler,\n            pos_weight,\n            epoch_seed=SEED + 1000 * mi + epoch,\n        )\n\n        val_pred = predict_member_frontier(\n            model, cache, slot_mask, val_idx, dev, IMG\n        )\n        val_bce = prob_bce(val_pred, y[val_idx], label_mask[val_idx])\n        val_auc, _ = macro_auc(val_pred, y[val_idx], label_mask[val_idx])\n\n        accepted = better_candidate(\n            val_auc, val_bce, best_auc, best_bce, parent_bce\n        )\n\n        history.append({\n            \"epoch\": epoch,\n            \"train_loss\": train_loss,\n            \"val_bce\": val_bce,\n            \"val_macro_auc\": val_auc,\n            \"accepted_by_gate\": bool(accepted),\n            \"head_lr\": optimizer.param_groups[0][\"lr\"],\n            \"backbone_lr\": optimizer.param_groups[1][\"lr\"],\n        })\n\n        log(\n            f\"  epoch {epoch}: train={train_loss:.5f}, \"\n            f\"val_bce={val_bce:.5f}, val_auc={val_auc:.4f}, \"\n            f\"gate={'ACCEPT' if accepted else 'reject'}\"\n        )\n\n        if accepted:\n            best_bce = val_bce\n            best_auc = val_auc\n            best_epoch = epoch\n            best_state = cpu_state_dict(model)\n            best_pred = val_pred.copy()\n            stale = 0\n        else:\n            stale += 1\n\n        scheduler.step()\n\n        if stale >= HFT_PATIENCE:\n            log(\n                f\"  early stop after epoch {epoch}; \"\n                f\"no gated improvement for {stale} epoch(s)\"\n            )\n            break\n\n    promoted = best_epoch > 0\n\n    if promoted:\n        model.load_state_dict(best_state, strict=True)\n    else:\n        model.load_state_dict(ck[\"model\"], strict=True)\n        best_bce = parent_bce\n        best_auc = parent_auc\n        best_pred = parent_pred.copy()\n\n    # FT-B gets its own deterministic fingerprint and records the seed.\n    new_fp_seed = int(SEED)\n    new_fp = fingerprint(model, dev, IMG, seed=new_fp_seed).tolist()\n    final_state = cpu_state_dict(model)\n\n    root_id = str(m.get(\"parent_id\") or m.get(\"id\")).replace(\"_fta\", \"\")\n    new_id = f\"{root_id}_ftb\"\n    new_file = f\"ftb_{root_id}.pt\"\n\n    new_ck = dict(ck)\n    new_ck[\"model\"] = final_state\n    new_ck[\"fingerprint\"] = new_fp\n    new_ck[\"fingerprint_seed\"] = int(new_fp_seed)\n    new_ck[\"ft_stage\"] = \"FT-B_HARD_EXAMPLE\"\n    new_ck[\"parent_stage\"] = \"FT-A_CONSERVATIVE\"\n    new_ck[\"parent_id\"] = m[\"id\"]\n    new_ck[\"parent_file\"] = m[\"file\"]\n    new_ck[\"root_base_id\"] = root_id\n    new_ck[\"ft_validation_mode\"] = \"NEW_EXPERT_CV_NOT_HISTORICAL_OOF\"\n    new_ck[\"ft_val_fold\"] = val_fold\n    new_ck[\"ft_unfreeze_last\"] = HFT_UNFREEZE_LAST\n    new_ck[\"ft_best_epoch\"] = int(best_epoch)\n    new_ck[\"ft_promoted\"] = bool(promoted)\n    new_ck[\"ft_parent_val_bce\"] = float(parent_bce)\n    new_ck[\"ft_best_val_bce\"] = float(best_bce)\n    new_ck[\"ft_parent_val_macro_auc\"] = (\n        float(parent_auc) if np.isfinite(parent_auc) else None\n    )\n    new_ck[\"ft_best_val_macro_auc\"] = (\n        float(best_auc) if np.isfinite(best_auc) else None\n    )\n    new_ck[\"hard_example_strategy\"] = {\n        \"source\": \"fold-local FT-A parent predictions on TRAINING partition only\",\n        \"validation_fold_excluded_from_mining\": True,\n        \"hard_boost\": HFT_HARD_BOOST,\n        \"draw_multiplier\": HFT_DRAW_MULT,\n        \"hard_quantile\": HFT_HARD_QUANTILE,\n        \"auc_first_gate\": True,\n    }\n    new_ck[\"ft_history\"] = history\n\n    torch.save(new_ck, OUT / new_file)\n\n    nm = dict(m)\n    nm[\"id\"] = new_id\n    nm[\"file\"] = new_file\n    nm[\"ft_stage\"] = \"FT-B_HARD_EXAMPLE\"\n    nm[\"parent_stage\"] = \"FT-A_CONSERVATIVE\"\n    nm[\"parent_id\"] = m[\"id\"]\n    nm[\"parent_file\"] = m[\"file\"]\n    nm[\"root_base_id\"] = root_id\n    nm[\"ft_promoted\"] = bool(promoted)\n    nm[\"ft_best_epoch\"] = int(best_epoch)\n    nm[\"ft_parent_val_bce\"] = float(parent_bce)\n    nm[\"ft_best_val_bce\"] = float(best_bce)\n    nm[\"ft_parent_val_macro_auc\"] = (\n        float(parent_auc) if np.isfinite(parent_auc) else None\n    )\n    nm[\"ft_best_val_macro_auc\"] = (\n        float(best_auc) if np.isfinite(best_auc) else None\n    )\n    new_members.append(nm)\n\n    for local_pos, cache_idx in enumerate(val_idx):\n        uid = studies[int(cache_idx)]\n        g = uid_to_global[uid]\n        oof_ftb_sum[g] += best_pred[local_pos]\n        oof_count[g] += 1\n\n    metrics_rows.append({\n        \"parent_id\": m[\"id\"],\n        \"new_id\": new_id,\n        \"base_fold\": int(m.get(\"fold\")),\n        \"ft_val_fold\": val_fold,\n        \"train_n\": len(train_idx),\n        \"val_n\": len(val_idx),\n        \"hard_n\": int(hard[\"is_hard\"].sum()),\n        \"mean_train_hard_weight\": float(study_weight.mean()),\n        \"parent_val_bce\": parent_bce,\n        \"best_val_bce\": best_bce,\n        \"delta_bce\": best_bce - parent_bce,\n        \"parent_val_macro_auc\": parent_auc,\n        \"best_val_macro_auc\": best_auc,\n        \"delta_auc\": best_auc - parent_auc if np.isfinite(parent_auc) else np.nan,\n        \"best_epoch\": best_epoch,\n        \"promoted\": promoted,\n        \"parent_oof_maxdiff\": parent_oof_maxdiff,\n        \"parent_fingerprint_seed\": int(parent_fp_seed) if \"fingerprint\" in ck else None,\n        \"new_fingerprint_seed\": int(new_fp_seed),\n    })\n\n    del model, ck, final_state, best_state, train_parent_pred\n    gc.collect()\n    if dev.type == \"cuda\":\n        torch.cuda.empty_cache()\n\nmetrics = pd.DataFrame(metrics_rows)\nmetrics.to_csv(OUT / \"ft_b_member_metrics.csv\", index=False)\ndisplay(metrics)\n\nhard_by_member = pd.DataFrame(hard_rows)\nhard_by_member.to_csv(OUT / \"hard_example_scores_by_member.csv\", index=False)\n\nif len(hard_by_member):\n    hard_summary = (\n        hard_by_member.groupby(\"StudyInstanceUID\", as_index=False)\n        .agg(\n            appearances=(\"parent_id\", \"count\"),\n            mean_hardness=(\"hardness\", \"mean\"),\n            max_hardness=(\"hardness\", \"max\"),\n            mean_sample_weight=(\"sample_weight\", \"mean\"),\n            hard_votes=(\"is_hard\", \"sum\"),\n        )\n        .sort_values([\"hard_votes\", \"mean_hardness\"], ascending=False)\n    )\nelse:\n    hard_summary = pd.DataFrame()\n\nhard_summary.to_csv(OUT / \"hard_example_summary.csv\", index=False)\ndisplay(hard_summary.head(20))","metadata":{"execution":{"iopub.status.busy":"2026-08-11T20:33:41.906452Z","iopub.execute_input":"2026-08-11T20:33:41.9067Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 8. Stage-level OOF diagnostic and FT-B handoff","metadata":{}},{"cell_type":"code","source":"valid = oof_count > 0\nif not valid.all():\n    missing_uids = expert.loc[~valid, \"StudyInstanceUID\"].tolist()\n    print(\"WARNING: some expert studies received no FT-B validation prediction:\", len(missing_uids))\n\nftb_oof = np.full_like(oof_ftb_sum, np.nan)\nftb_oof[valid] = oof_ftb_sum[valid] / oof_count[valid, None]\n\nbase_oof = np.column_stack([oof2[f\"base__{t}\"].to_numpy(float) for t in TARGETS])\nfta_oof = np.column_stack([oof2[f\"fta__{t}\"].to_numpy(float) for t in TARGETS])\n\noof_frame = expert[[\"StudyInstanceUID\", \"ft_fold\", *TARGETS]].copy()\noof_frame[\"n_ftb_votes\"] = oof_count\n\nfor j, t in enumerate(TARGETS):\n    oof_frame[f\"base__{t}\"] = base_oof[:, j]\n    oof_frame[f\"fta__{t}\"] = fta_oof[:, j]\n    oof_frame[f\"ftb__{t}\"] = ftb_oof[:, j]\n\noof_frame.to_csv(OUT / \"expert_cv_predictions_stage3.csv\", index=False)\n\ny_all = expert[TARGETS].to_numpy(np.float32)\nmask_all = np.isfinite(y_all).astype(np.float32)\ny_safe = np.nan_to_num(y_all, nan=0.0)\n\ndef branch_metrics(name, pred):\n    bce = prob_bce(pred[valid], y_safe[valid], mask_all[valid])\n    auc, per = macro_auc(pred[valid], y_safe[valid], mask_all[valid])\n    return name, bce, auc, per\n\nbranches = [\n    branch_metrics(\"BASE\", base_oof),\n    branch_metrics(\"FT-A\", fta_oof),\n    branch_metrics(\"FT-B\", ftb_oof),\n]\n\nsummary = pd.DataFrame([\n    {\"branch\": name, \"expert_cv_bce\": bce, \"expert_cv_macro_auc\": auc}\n    for name, bce, auc, _ in branches\n]).sort_values(\"expert_cv_macro_auc\", ascending=False)\n\nsummary.to_csv(OUT / \"expert_cv_branch_summary.csv\", index=False)\ndisplay(summary)\n\nper_target = pd.DataFrame({\"target\": TARGETS})\nfor name, _, _, per in branches:\n    per_target[f\"{name.lower().replace('-', '')}_auc\"] = [per[t] for t in TARGETS]\n\nper_target[\"ftb_minus_fta\"] = per_target[\"ftb_auc\"] - per_target[\"fta_auc\"]\nper_target[\"ftb_minus_base\"] = per_target[\"ftb_auc\"] - per_target[\"base_auc\"]\nper_target.to_csv(OUT / \"expert_cv_target_auc_stage3.csv\", index=False)\ndisplay(per_target)\n\nbase_bce = float(summary.loc[summary[\"branch\"] == \"BASE\", \"expert_cv_bce\"].iloc[0])\nfta_bce = float(summary.loc[summary[\"branch\"] == \"FT-A\", \"expert_cv_bce\"].iloc[0])\nftb_bce = float(summary.loc[summary[\"branch\"] == \"FT-B\", \"expert_cv_bce\"].iloc[0])\nbase_auc = float(summary.loc[summary[\"branch\"] == \"BASE\", \"expert_cv_macro_auc\"].iloc[0])\nfta_auc = float(summary.loc[summary[\"branch\"] == \"FT-A\", \"expert_cv_macro_auc\"].iloc[0])\nftb_auc = float(summary.loc[summary[\"branch\"] == \"FT-B\", \"expert_cv_macro_auc\"].iloc[0])\n\nbest_single = str(\n    summary.sort_values(\n        [\"expert_cv_macro_auc\", \"expert_cv_bce\"],\n        ascending=[False, True],\n    ).iloc[0][\"branch\"]\n)\n\nhistory = stage2_manifest.get(\"history\")\nif not isinstance(history, list):\n    history = [] if history is None else [history]\n\nstage3_history = history + [{\n    \"stage\": \"FT-B_HARD_EXAMPLE\",\n    \"seed\": SEED,\n    \"fingerprint_seed\": SEED,\n    \"parent_fingerprint_seed\": STAGE2_FINGERPRINT_SEED,\n    \"n_output_members\": len(new_members),\n    \"expert_n\": len(expert),\n    \"validation_mode\": \"NEW_EXPERT_CV_NOT_HISTORICAL_OOF\",\n    \"hardness_source\": \"fold-local FT-A parent TRAINING predictions only\",\n    \"unfreeze_last\": HFT_UNFREEZE_LAST,\n    \"max_epochs\": HFT_MAX_EPOCHS,\n    \"head_lr\": HFT_HEAD_LR,\n    \"backbone_lr\": HFT_BACKBONE_LR,\n    \"expert_cv_base_bce\": base_bce,\n    \"expert_cv_fta_bce\": fta_bce,\n    \"expert_cv_ftb_bce\": ftb_bce,\n    \"expert_cv_base_macro_auc\": base_auc,\n    \"expert_cv_fta_macro_auc\": fta_auc,\n    \"expert_cv_ftb_macro_auc\": ftb_auc,\n    \"best_single_cv_branch_by_macro_auc\": best_single,\n}]\n\nnew_manifest = {\n    \"format\": stage2_manifest.get(\"format\"),\n    \"targets\": TARGETS,\n    \"history\": stage3_history,\n    \"members\": new_members,\n}\n(OUT / \"manifest.json\").write_text(json.dumps(new_manifest, indent=2))\n\nstatus3 = {\n    \"stage3_ok\": len(new_members) == len(members) and len(new_members) > 0,\n    \"stage\": \"FT-B_HARD_EXAMPLE\",\n    \"fingerprint_seed\": int(SEED),\n    \"parent_fingerprint_seed\": int(STAGE2_FINGERPRINT_SEED),\n    \"n_input_members\": len(members),\n    \"n_output_members\": len(new_members),\n    \"n_promoted\": int(metrics[\"promoted\"].sum()) if len(metrics) else 0,\n    \"n_rolled_back_to_fta\": int((~metrics[\"promoted\"]).sum()) if len(metrics) else 0,\n    \"validation_mode\": \"NEW_EXPERT_CV_NOT_HISTORICAL_OOF\",\n    \"hardness_source\": \"FOLD_LOCAL_FTA_PARENT_TRAIN_PREDICTIONS_ONLY\",\n    \"expert_cv_base_bce\": base_bce,\n    \"expert_cv_fta_bce\": fta_bce,\n    \"expert_cv_ftb_bce\": ftb_bce,\n    \"expert_cv_base_macro_auc\": base_auc,\n    \"expert_cv_fta_macro_auc\": fta_auc,\n    \"expert_cv_ftb_macro_auc\": ftb_auc,\n    \"best_single_cv_branch_by_macro_auc\": best_single,\n    \"submission_file_policy\": (\n        \"No interim submission file. Final notebook writes exactly \"\n        \"/kaggle/working/submission.csv after OOF ensemble selection.\"\n    ),\n    \"next_notebook\": \"04_target_specialist_finetune.ipynb\",\n}\n(OUT / \"stage3_status.json\").write_text(json.dumps(status3, indent=2))\n\nprint(json.dumps(status3, indent=2))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 9. Output inventory","metadata":{}},{"cell_type":"code","source":"print(\"Stage 3 output:\")\ntotal_bytes = 0\nfor p in sorted(OUT.iterdir()):\n    if p.is_file():\n        total_bytes += p.stat().st_size\n        print(f\" - {p.name:42s} {p.stat().st_size / (1024**2):9.2f} MB\")\n\nprint(f\"Total output size: {total_bytes / (1024**3):.3f} GB\")\nprint()\nprint(\"IMPORTANT: no submission.csv is created in Stage 3.\")\nprint(\"Next: add this Notebook 03 output as a Kaggle input to Notebook 04.\")\nprint(\"Final submission stage will write exactly /kaggle/working/submission.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## What to send back after the run\n\nPlease share:\n\n- `stage3_status.json`\n- `expert_cv_branch_summary.csv`\n- `expert_cv_target_auc_stage3.csv`\n- the final output/log screenshot\n\nThe next notebook will be **FT-C target-specialist fine-tuning**.  \nAfter FT-C we will compare BASE vs FT-A vs FT-B vs FT-C using the same expert-CV evidence before building the final ensemble.","metadata":{}}]}