{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":118765,"databundleVersionId":15231210,"sourceType":"competition"},{"sourceId":11775065,"sourceType":"datasetVersion","datasetId":7392749},{"sourceId":11788894,"sourceType":"datasetVersion","datasetId":7402173},{"sourceId":291833135,"sourceType":"kernelVersion"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":413.683059,"end_time":"2026-01-15T23:01:20.07","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-01-15T22:54:26.386941","version":"2.6.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"06627006a5d1437ba85aad3b90e50fa1":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"1375cd19c78c465385f7abc7a4897ebf":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"182e26be764c4ffe8e844edec85f8fca":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_19d6f36f2b9d4f388a2eed93129cd644","placeholder":"​","style":"IPY_MODEL_402ab8db52ff4f25b06868f88d334aba","tabbable":null,"tooltip":null,"value":" 5716/5716 [01:11&lt;00:00, 58.27it/s]"}},"19d6f36f2b9d4f388a2eed93129cd644":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"1a5937184aaf4725abaa25e8ffceee95":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"287a83750ffd4a3399bcde35ff4cc0cd":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"2e244c81200649d7ae41b80a18ddc818":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"2e8b4a7445a843e5afa10f321ed7da7d":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"402ab8db52ff4f25b06868f88d334aba":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"4a7e8fd65b1f4e3b9f9ac68879a1ee55":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"4bda594d0ff742618121123d2c657c66":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"51e295a2120b4a24810a9f6128317971":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_803b9861678641b89902efbd5b861a38","max":5716,"min":0,"orientation":"horizontal","style":"IPY_MODEL_2e244c81200649d7ae41b80a18ddc818","tabbable":null,"tooltip":null,"value":5716}},"55f5c4a495f1482ea9d47e35fec64650":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"5975a39a4cf343f68079032889a85b5e":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"6492fd50eda240008e47bfdcaf84a4a1":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"70aff8b3e5374ae39fb8427a231afc48":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"7dc17d96e0984072999e36ec2addfdc1":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_e0c1cc7aeee14b40a757ba3b5dce9289","max":28,"min":0,"orientation":"horizontal","style":"IPY_MODEL_1375cd19c78c465385f7abc7a4897ebf","tabbable":null,"tooltip":null,"value":28}},"7dd76bc86f944ce4a95bcd581b4dfd0a":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"803b9861678641b89902efbd5b861a38":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"81d6d1f4a8d34dab83dea9a497f01eab":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_7dd76bc86f944ce4a95bcd581b4dfd0a","placeholder":"​","style":"IPY_MODEL_4a7e8fd65b1f4e3b9f9ac68879a1ee55","tabbable":null,"tooltip":null,"value":"Processing structures: 100%"}},"86bae1783d314095b4a765c3820f422b":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8d47b18625a44ffa842907a197dacbaf":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_55f5c4a495f1482ea9d47e35fec64650","placeholder":"​","style":"IPY_MODEL_5975a39a4cf343f68079032889a85b5e","tabbable":null,"tooltip":null,"value":"100%"}},"90eeb3fb273e4ea9a2d41471eaaa2186":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"a1fabf3e16884b9f8cc08c907f0802fe":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_81d6d1f4a8d34dab83dea9a497f01eab","IPY_MODEL_51e295a2120b4a24810a9f6128317971","IPY_MODEL_e04e0b830cdc455c9d293f366b4dc729"],"layout":"IPY_MODEL_86bae1783d314095b4a765c3820f422b","tabbable":null,"tooltip":null}},"a7c5d90c4a4a4ae093014ce9ff16394b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_06627006a5d1437ba85aad3b90e50fa1","max":5716,"min":0,"orientation":"horizontal","style":"IPY_MODEL_1a5937184aaf4725abaa25e8ffceee95","tabbable":null,"tooltip":null,"value":5716}},"b2474bbbb4f245bf99e383ca5f8818ea":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"cc20cd5bc78f4a42949f3ebb8ea50d61":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_e98360dd1d2a49d1af4efba45d0a13cb","placeholder":"​","style":"IPY_MODEL_2e8b4a7445a843e5afa10f321ed7da7d","tabbable":null,"tooltip":null,"value":" 28/28 [01:18&lt;00:00,  2.49s/it]"}},"d39e56543eb64f76813d53d305fe58fb":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_f799e72583fe4b6a95750b64afe59f75","IPY_MODEL_7dc17d96e0984072999e36ec2addfdc1","IPY_MODEL_cc20cd5bc78f4a42949f3ebb8ea50d61"],"layout":"IPY_MODEL_4bda594d0ff742618121123d2c657c66","tabbable":null,"tooltip":null}},"e04e0b830cdc455c9d293f366b4dc729":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_6492fd50eda240008e47bfdcaf84a4a1","placeholder":"​","style":"IPY_MODEL_90eeb3fb273e4ea9a2d41471eaaa2186","tabbable":null,"tooltip":null,"value":" 5716/5716 [00:06&lt;00:00, 1001.01it/s]"}},"e0c1cc7aeee14b40a757ba3b5dce9289":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"e98360dd1d2a49d1af4efba45d0a13cb":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"f41bca461ce1440f92dcaf8ebbd9379a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_8d47b18625a44ffa842907a197dacbaf","IPY_MODEL_a7c5d90c4a4a4ae093014ce9ff16394b","IPY_MODEL_182e26be764c4ffe8e844edec85f8fca"],"layout":"IPY_MODEL_70aff8b3e5374ae39fb8427a231afc48","tabbable":null,"tooltip":null}},"f799e72583fe4b6a95750b64afe59f75":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_b2474bbbb4f245bf99e383ca5f8818ea","placeholder":"​","style":"IPY_MODEL_287a83750ffd4a3399bcde35ff4cc0cd","tabbable":null,"tooltip":null,"value":"100%"}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Stanford RNA 3D Folding Part 2 - Templates\n\n**About:**\nGenerate templates for the Stanford RNA v2 competition using 2 approches:\n- MMSEQ\n- TBM only by @aejohn, which won lthe previous iteration of the competition\n\n**Improvements:**\n- Use `parasail` for faster distance computation\n- Rewrite `process_labels` more efficiently\n- Precompute `train_features` instead of computing them at each iteration","metadata":{"papermill":{"duration":0.007908,"end_time":"2026-01-15T22:54:29.58142","exception":false,"start_time":"2026-01-15T22:54:29.573512","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!pip install /kaggle/input/parasail/biopython-1.85-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl --no-deps\n!pip install /kaggle/input/parasail/parasail-1.3.4-py2.py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl --no-deps","metadata":{"_cell_guid":"78d00d9d-393a-4f19-af45-d6d07a523490","_kg_hide-input":true,"_kg_hide-output":true,"_uuid":"7ac2863a-6552-4372-a3aa-9900b3d4255a","collapsed":false,"execution":{"iopub.execute_input":"2026-01-15T22:54:29.598304Z","iopub.status.busy":"2026-01-15T22:54:29.597939Z","iopub.status.idle":"2026-01-15T22:54:36.85356Z","shell.execute_reply":"2026-01-15T22:54:36.852435Z"},"jupyter":{"outputs_hidden":false},"papermill":{"duration":7.266548,"end_time":"2026-01-15T22:54:36.855851","exception":false,"start_time":"2026-01-15T22:54:29.589303","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Generate templates with MMSEQ\n- Using https://www.kaggle.com/code/rhijudas/mmseqs2-3d-rna-template-identification","metadata":{"papermill":{"duration":0.006941,"end_time":"2026-01-15T22:54:36.870067","exception":false,"start_time":"2026-01-15T22:54:36.863126","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport csv\nimport gzip\nimport sys\nimport warnings\nimport argparse\nfrom datetime import datetime\n\nfrom Bio import SeqIO,PDB,BiopythonWarning\nfrom Bio.PDB.MMCIF2Dict import MMCIF2Dict\nfrom Bio.Seq import Seq\nfrom Bio.PDB import MMCIFParser\n\n# Suppress warnings\nwarnings.simplefilter('ignore', BiopythonWarning)","metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2026-01-15T22:54:36.886742Z","iopub.status.busy":"2026-01-15T22:54:36.886348Z","iopub.status.idle":"2026-01-15T22:54:37.95031Z","shell.execute_reply":"2026-01-15T22:54:37.949542Z"},"papermill":{"duration":1.075027,"end_time":"2026-01-15T22:54:37.95237","exception":false,"start_time":"2026-01-15T22:54:36.877343","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Utils","metadata":{"papermill":{"duration":0.006983,"end_time":"2026-01-15T22:54:37.966439","exception":false,"start_time":"2026-01-15T22:54:37.959456","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def clean_res_name(res_name):\n    if res_name in [\"A\", \"C\", \"G\", \"U\"]:\n        return res_name\n    else:  # can be modified residue with 3-letter name.\n        return \"X\"\n\n\ndef extract_title_release_date(cif_path):\n\n    if cif_path.endswith(\".gz\"):\n        with gzip.open(cif_path, \"rt\") as cif_file:\n            mmcif_dict = MMCIF2Dict(cif_file)\n    else:\n        mmcif_dict = MMCIF2Dict(cif_path)\n\n    possible_title_fields = [\n        \"_struct.title\",\n        \"_entry.title\",\n        \"_struct_keywords.pdbx_keywords\",\n    ]\n\n    pdb_title = None\n    for field in possible_title_fields:\n        if field in mmcif_dict:\n            pdb_title = mmcif_dict[field]\n            if isinstance(pdb_title, list):\n                pdb_title = \" \".join(pdb_title)\n            break\n\n    possible_date_fields = [\n        \"_pdbx_database_status.initial_release_date\",\n        \"_pdbx_database_status.recvd_initial_deposition_date\",\n        \"_database_PDB_rev.date\",\n    ]\n\n    release_date = None\n    for field in possible_date_fields:\n        if field in mmcif_dict:\n            release_date = mmcif_dict[field]\n            if isinstance(release_date, list):\n                release_date = release_date[0]  # Take the first date if it's a list\n            break\n\n    return pdb_title, release_date\n\n\ndef extract_rna_sequence(cif_path, chain_id):\n\n    if cif_path.endswith(\".gz\"):\n        with gzip.open(cif_path, \"rt\") as cif_file:\n            mmcif_dict = MMCIF2Dict(cif_file)\n    else:\n        mmcif_dict = MMCIF2Dict(cif_path)\n\n    pdb_sequence = None\n    pdb_chain_id = None\n    chain_seq_nums = None\n\n    # Extract _pdbx_poly_seq_scheme information\n    strand_id = mmcif_dict.get(\"_pdbx_poly_seq_scheme.pdb_strand_id\", [])\n    mon_id = mmcif_dict.get(\"_pdbx_poly_seq_scheme.mon_id\", [])\n    pdb_mon_id = mmcif_dict.get(\"_pdbx_poly_seq_scheme.pdb_mon_id\", [])\n    pdb_seq_num = mmcif_dict.get(\"_pdbx_poly_seq_scheme.pdb_seq_num\", [])\n    chain_ids = list(set(strand_id))\n    seq_chains = []\n\n    full_sequence = \"\"\n    pdb_chain_sequence = \"\"\n    pdb_chain_seq_nums = []\n    for strand, mon, pdb_mon, pdb_num in zip(\n        strand_id, mon_id, pdb_mon_id, pdb_seq_num\n    ):\n        if strand == chain_id:\n            full_sequence += clean_res_name(mon)\n            pdb_chain_sequence += clean_res_name(pdb_mon)\n            pdb_chain_seq_nums.append(pdb_num)\n\n    # print(full_sequence)\n    # print(pdb_chain_sequence)\n    # print(pdb_chain_seq_nums)\n\n    return full_sequence, pdb_chain_sequence, pdb_chain_seq_nums\n\n\ndef get_c1prime_labels(cif_path, chain_id, alignment, chain_seq_nums):\n    \"\"\"\n    Extract C1' coordinates for an RNA chain based on a reference sequence alignment.\n\n    This function uses Biopython to parse a CIF file, finds the specified chain,\n    and extracts C1' coordinates for RNA residues. It aligns these coordinates\n    with a reference sequence, handling gaps and missing residues.\n\n    Parameters:\n    cif_path (str): Path to the CIF file.\n    chain_id (str): Chain identifier in the CIF file.\n    alignment (list): A list containing two elements:\n                      alignment[0]: List of residues for the reference sequence (A,C,G,U,-)\n                      alignment[1]: List of residues for the chain sequence (A,C,G,U,X,-)\n    chain_seq_nums (list): numbers of residues in PDB\n\n    Returns:\n    list of tuples: Each tuple contains (resname, resid, x, y, z), where:\n                    resname: Residue name (A, C, G, or U) from the reference sequence\n                    resid: Residue ID (1, 2, 3, ...) based on position in reference sequence\n                    x, y, z: C1' coordinates (nan or NULL_VALUE for missing residues/atoms)\n\n    The length of the returned list is equal to the number of non-gap residues\n    in the reference sequence.\n    \"\"\"\n    # Parse the CIF file\n    parser = MMCIFParser()\n    if cif_path.endswith(\".gz\"):\n        with gzip.open(cif_path, \"rt\") as gz_file:\n            structure = parser.get_structure(\"RNA\", gz_file)\n    else:\n        structure = parser.get_structure(\"RNA\", cif_path)\n\n    # Get the specified chain\n    chain = structure[0][chain_id]\n\n    # getting residues out of chain is complex -- easier to get a list ahead of time.\n    residues = {}\n    for residue in chain:\n        residues[residue.id[1]] = residue\n\n    chain_seq = \"\".join([clean_res_name(residue.get_resname()) for residue in chain])\n    # print(chain_seq)\n\n    # Initialize the result list\n    result = []\n\n    # Counter for residue ID in reference sequence\n    ref_resid = 0\n    chain_idx = 0\n    for ref_res, chain_res in zip(alignment[0], alignment[1]):\n        if chain_res != \"-\":\n            chain_idx += 1\n        if ref_res != \"-\":\n            ref_resid += 1\n            if chain_res == \"-\":  # or chain_res == 'X':\n                # Missing residue in chain or unknown residue\n                result.append(\n                    (ref_res, ref_resid, NULL_VALUE, NULL_VALUE, NULL_VALUE, -1e18)\n                )\n            else:\n                # Find the corresponding residue in the chain\n                try:\n                    # chain_seq_num = int(chain_seq_nums[ref_resid-1])\n                    chain_seq_num = int(chain_seq_nums[chain_idx - 1])\n                    residue = residues[chain_seq_num]\n                    c1_prime = residue[\"C1'\"]\n                    coords = c1_prime.coord\n                    if residue.get_resname() != chain_res:\n                        print(\n                            \"WARNING!\",\n                            ref_resid,\n                            chain_idx,\n                            chain_seq_num,\n                            residue.get_resname(),\n                            chain_res,\n                        )\n                    result.append(\n                        (\n                            ref_res,\n                            ref_resid,\n                            coords[0],\n                            coords[1],\n                            coords[2],\n                            residue.id[1],\n                        )\n                    )\n                except KeyError:\n                    # C1' atom not found\n                    result.append(\n                        (ref_res, ref_resid, NULL_VALUE, NULL_VALUE, NULL_VALUE, -1e18)\n                    )\n                except Exception as e:\n                    # Any other error (e.g., residue not found)\n                    result.append(\n                        (ref_res, ref_resid, NULL_VALUE, NULL_VALUE, NULL_VALUE, -1e18)\n                    )\n\n    return result\n\n\ndef is_before_or_on(d1, d2):\n    date1 = pd.to_datetime(d1)\n    date2 = pd.to_datetime(d2)\n    return date1 <= date2\n\n\ndef read_id_map(id_map_file):\n    if len(id_map_file) == 0:\n        return None\n    id_map = {}\n    try:\n        with open(id_map_file, newline=\"\") as f:\n            reader = csv.DictReader(f)\n            if \"orig\" not in reader.fieldnames or \"new\" not in reader.fieldnames:\n                print(\n                    \"Warning: ID map file does not contain the fields 'orig' and 'new'. Using original IDs instead.\"\n                )\n                return id_map\n            for row in reader:\n                id_map[row[\"orig\"]] = row[\"new\"]\n    except FileNotFoundError:\n        print(\n            f\"Warning: ID map file {id_map_file} not found. Using original IDs instead.\",\n            file=sys.stderr,\n        )\n    except Exception as exc:\n        print(f\"Error reading {id_map_file}: {exc}\", file=sys.stderr)\n    return id_map\n\n\ndef read_release_dates(release_data_file):\n    release_dates = {}\n    # must have format Entry ID, Release Date\n    with open(release_data_file, newline=\"\") as f:\n        reader = csv.DictReader(f)\n        for row in reader:\n            release_dates[row[\"Entry ID\"]] = row[\"Release Date\"]\n\n    return release_dates\n\n\n# Create a DataFrame and write to CSV\ndef output_csv(output_data, outfile):\n    df = pd.DataFrame(output_data)\n    df.to_csv(outfile, index=False)\n    print(f\"Output written to {outfile}\")","metadata":{"_cell_guid":"3922db17-a650-4f75-b073-46d2767a0c93","_kg_hide-input":true,"_uuid":"6a9568b3-922b-4d41-8b10-39b2678e9b71","execution":{"iopub.execute_input":"2026-01-15T22:54:37.98343Z","iopub.status.busy":"2026-01-15T22:54:37.982784Z","iopub.status.idle":"2026-01-15T22:54:38.006164Z","shell.execute_reply":"2026-01-15T22:54:38.005263Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.033833,"end_time":"2026-01-15T22:54:38.007994","exception":false,"start_time":"2026-01-15T22:54:37.974161","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Params","metadata":{"papermill":{"duration":0.006897,"end_time":"2026-01-15T22:54:38.022277","exception":false,"start_time":"2026-01-15T22:54:38.01538","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# if False, don't check on temporal_cutoff -- during training on known structures, will get lots of leakage. \n# If going for Early Sharing Prize, set to True.\nCHECK_TEMPORAL_CUTOFF = True\n\n# Number of templates to use. \n# Here set to 5 to prepare a mock submission\n# Should make larger (e.g., 40) if using templates for modeling. \nMAX_TEMPLATES = 5\n\n# Better to use nan when preparing files for templates, to allow easy recognition of which coordinates are missing.\n# But for this example, using 0.0 to avoid errors in scoring the final submission.csv\nNULL_VALUE = 0.0\n\nCOMP = \"/kaggle/input/stanford-rna-3d-folding-2/\"","metadata":{"_cell_guid":"a49598c1-eb79-4f53-801d-b7b368221963","_uuid":"fc10b6af-500d-4119-b0e0-32cddfa0eb60","collapsed":false,"execution":{"iopub.execute_input":"2026-01-15T22:54:38.037791Z","iopub.status.busy":"2026-01-15T22:54:38.037433Z","iopub.status.idle":"2026-01-15T22:54:38.042171Z","shell.execute_reply":"2026-01-15T22:54:38.041293Z"},"jupyter":{"outputs_hidden":false},"papermill":{"duration":0.014874,"end_time":"2026-01-15T22:54:38.044122","exception":false,"start_time":"2026-01-15T22:54:38.029248","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Download MMseqs2","metadata":{"_cell_guid":"f8e0f6f8-43de-4c75-b588-58b05a0df05b","_uuid":"1f0c9549-ab5e-4375-af61-a42f8b1eee63","collapsed":false,"jupyter":{"outputs_hidden":false},"papermill":{"duration":0.007011,"end_time":"2026-01-15T22:54:38.058111","exception":false,"start_time":"2026-01-15T22:54:38.0511","status":"completed"},"tags":[]}},{"cell_type":"code","source":"#!wget https://mmseqs.com/latest/mmseqs-linux-avx2.tar.gz\n#!tar xvfz /kaggle/working/mmseqs-linux-avx2.tar.gz\n!rsync -avL /kaggle/input/mmseqs2/mmseqs /kaggle/working/\n!chmod 755 /kaggle/working/mmseqs/bin/mmseqs","metadata":{"_cell_guid":"57913cbf-9d04-4cf4-b7ea-48e47815224d","_kg_hide-input":false,"_kg_hide-output":true,"_uuid":"08b31882-2f49-4c30-85cd-8532e8a1ae96","collapsed":false,"execution":{"iopub.execute_input":"2026-01-15T22:54:38.073869Z","iopub.status.busy":"2026-01-15T22:54:38.073006Z","iopub.status.idle":"2026-01-15T22:54:39.621253Z","shell.execute_reply":"2026-01-15T22:54:39.620051Z"},"jupyter":{"outputs_hidden":false},"papermill":{"duration":1.558516,"end_time":"2026-01-15T22:54:39.623416","exception":false,"start_time":"2026-01-15T22:54:38.0649","status":"completed"},"scrolled":true,"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Create DB based on FASTA of all PDB nucleic acid sequences, which is part of PDB_RNA dataset","metadata":{"_cell_guid":"0d2f5d97-ab63-4f8a-a4ce-3d0fbb2bc8d1","_uuid":"0f0c130a-477a-484e-b607-1e830d322d21","collapsed":false,"jupyter":{"outputs_hidden":false},"papermill":{"duration":0.007469,"end_time":"2026-01-15T22:54:39.638795","exception":false,"start_time":"2026-01-15T22:54:39.631326","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!/kaggle/working/mmseqs/bin/mmseqs createdb $COMP/PDB_RNA/pdb_seqres_NA.fasta pdb_seqres_NA --dbtype 2 # > MMseqs_createDB.log","metadata":{"_cell_guid":"725fc627-e1d5-44cc-b96a-cfc01716217a","_uuid":"f7469ab6-c448-4e39-a8c2-ff9df635187a","collapsed":false,"execution":{"iopub.execute_input":"2026-01-15T22:54:39.656283Z","iopub.status.busy":"2026-01-15T22:54:39.655498Z","iopub.status.idle":"2026-01-15T22:54:40.05821Z","shell.execute_reply":"2026-01-15T22:54:40.057197Z"},"jupyter":{"outputs_hidden":false},"papermill":{"duration":0.413997,"end_time":"2026-01-15T22:54:40.060311","exception":false,"start_time":"2026-01-15T22:54:39.646314","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Find templates for targets in test_sequences.csv by aligning to PDB with MMseqs2\n\nNeed to  convert test_sequences.csv file to FASTA","metadata":{"_cell_guid":"3591eed4-14c3-42f8-bc96-c0586817bc5a","_uuid":"de1bb0c5-28c1-46c0-92b9-a35b257fdd09","collapsed":false,"jupyter":{"outputs_hidden":false},"papermill":{"duration":0.008241,"end_time":"2026-01-15T22:54:40.076765","exception":false,"start_time":"2026-01-15T22:54:40.068524","status":"completed"},"tags":[]}},{"cell_type":"code","source":"input_file = COMP + \"test_sequences.csv\"\noutput_file = \"test_sequences.fasta\"\n\nwith open(input_file, \"r\", newline=\"\") as csv_file, open(\n    output_file, \"w\"\n) as fasta_file:\n    csv_reader = csv.reader(\n        csv_file,\n        quotechar='\"',\n        delimiter=\",\",\n        quoting=csv.QUOTE_ALL,\n        skipinitialspace=True,\n    )\n    next(csv_reader)  # Skip the header row\n    for row in csv_reader:\n        if len(row) >= 2:\n            fasta_file.write(f\">{row[0]}\\n{row[1]}\\n\")\n","metadata":{"_cell_guid":"a7316dce-366f-4b2c-a278-829fb0fbb910","_uuid":"b1b4c396-0b08-49d2-90ab-52b2c8a89d4c","collapsed":false,"execution":{"iopub.execute_input":"2026-01-15T22:54:40.094858Z","iopub.status.busy":"2026-01-15T22:54:40.094417Z","iopub.status.idle":"2026-01-15T22:54:40.107496Z","shell.execute_reply":"2026-01-15T22:54:40.106785Z"},"jupyter":{"outputs_hidden":false},"papermill":{"duration":0.024584,"end_time":"2026-01-15T22:54:40.109407","exception":false,"start_time":"2026-01-15T22:54:40.084823","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!/kaggle/working/mmseqs/bin/mmseqs easy-search /kaggle/working/test_sequences.fasta /kaggle/working/pdb_seqres_NA testResult.txt tmp --search-type 3 --format-output \"query,target,evalue,qstart,qend,tstart,tend,qaln,taln\"","metadata":{"_cell_guid":"5bc7c1a8-253c-471c-8712-2e3bf24ca703","_kg_hide-output":true,"_uuid":"bdf4834c-1320-4402-85ce-a6d55d18403a","execution":{"iopub.execute_input":"2026-01-15T22:54:40.127365Z","iopub.status.busy":"2026-01-15T22:54:40.12703Z","iopub.status.idle":"2026-01-15T22:54:57.145142Z","shell.execute_reply":"2026-01-15T22:54:57.143918Z"},"papermill":{"duration":17.029695,"end_time":"2026-01-15T22:54:57.147343","exception":false,"start_time":"2026-01-15T22:54:40.117648","status":"completed"},"scrolled":true,"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Setup","metadata":{"_cell_guid":"d6f776cf-d311-48cd-8b88-28e24685cad2","_uuid":"7423d507-4986-4731-a247-0c6d9bffee17","collapsed":false,"jupyter":{"outputs_hidden":false},"papermill":{"duration":0.012457,"end_time":"2026-01-15T22:54:57.172517","exception":false,"start_time":"2026-01-15T22:54:57.16006","status":"completed"},"tags":[]}},{"cell_type":"code","source":"sequences_file = COMP + 'test_sequences.csv'\nmmseqs_results_file = '/kaggle/working/testResult.txt' # created by MMseqs command lines\noutfile = 'templates_mmseq.csv'\ncif_dir = COMP + '/PDB_RNA'\n\n# variables not in use here:\nid_map_file = ''\nstart_idx = 0\nend_idx = 0","metadata":{"_cell_guid":"873c9e72-c31f-4f55-aa33-e86ddfca1195","_kg_hide-input":true,"_uuid":"6d7546da-2661-473e-823f-af08aa4be9d3","collapsed":false,"execution":{"iopub.execute_input":"2026-01-15T22:54:57.199085Z","iopub.status.busy":"2026-01-15T22:54:57.198703Z","iopub.status.idle":"2026-01-15T22:54:57.204165Z","shell.execute_reply":"2026-01-15T22:54:57.203403Z"},"jupyter":{"outputs_hidden":false},"papermill":{"duration":0.020994,"end_time":"2026-01-15T22:54:57.205954","exception":false,"start_time":"2026-01-15T22:54:57.18496","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare to collect output data\noutput_labels = []\n\n# Read the FASTA file\ndf = pd.read_csv(sequences_file)\ntargets = df[\"target_id\"].to_list()\nsequences = df[\"sequence\"].to_list()\ntemporal_cutoffs = df[\"temporal_cutoff\"].to_list()\n\naln_lines = []\nfor line in open(mmseqs_results_file).readlines():\n    # query,template,eval,qstart,qend,tstart,tend,qaln,taln\n    aln_lines.append(line.strip().split())\n\nid_map = read_id_map(id_map_file)\n\nrelease_dates = read_release_dates(cif_dir + \"/pdb_release_dates_NA.csv\")\n\nif start_idx == 0 and end_idx == 0:  # do all targets by default\n    start_idx = 1\n    end_idx = len(targets)","metadata":{"_cell_guid":"38601ec7-2f75-4d63-b2db-d5d03eccdb43","_uuid":"e7e66851-9830-466b-9435-ec3e1dda980c","collapsed":false,"execution":{"iopub.execute_input":"2026-01-15T22:54:57.232137Z","iopub.status.busy":"2026-01-15T22:54:57.231708Z","iopub.status.idle":"2026-01-15T22:54:57.288697Z","shell.execute_reply":"2026-01-15T22:54:57.287951Z"},"jupyter":{"outputs_hidden":false},"papermill":{"duration":0.072548,"end_time":"2026-01-15T22:54:57.290808","exception":false,"start_time":"2026-01-15T22:54:57.21826","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Assemble file of template coordinates by going through .cif files found for each target by MMseqs2","metadata":{"_cell_guid":"fa72dcfa-28a7-4a60-aef2-545ebd7b963d","_uuid":"520f380e-732a-48b3-b5d6-8f4e35a430fb","collapsed":false,"jupyter":{"outputs_hidden":false},"papermill":{"duration":0.012325,"end_time":"2026-01-15T22:54:57.315586","exception":false,"start_time":"2026-01-15T22:54:57.303261","status":"completed"},"tags":[]}},{"cell_type":"code","source":"num_targets = 0\ncount = 0\nfor target, sequence, temporal_cutoff in zip(targets, sequences, temporal_cutoffs):\n    count += 1\n    if (count < start_idx) or (count > end_idx):\n        continue\n\n    # look for alignments and fill out C1' templates\n    templates = []\n    for aln_line in aln_lines:\n        if len(aln_line) != 9:\n            continue  # some kind of overflow in some alignments?\n\n        query, template, eval, qstart, qend, tstart, tend, qaln, taln = aln_line\n\n        if query != target:\n            continue\n\n        if int(qend) < int(qstart):\n            continue  # aligned to reverse complement!\n\n        pdb_id, chain_id = template.split(\"_\")\n\n        # need to do alignment\n        cif_path = os.path.join(cif_dir, f\"{pdb_id.lower()}.cif\")\n        if not os.path.isfile(cif_path):\n            continue  # occasional alignment to DNA, ignore!\n\n        release_date = release_dates[pdb_id.upper()]  # pulled from PDB server\n\n        if CHECK_TEMPORAL_CUTOFF and is_before_or_on(temporal_cutoff, release_date):\n            continue\n\n        # these release dates in the CIF files can be buggy!\n        title, release_date_unreliable = extract_title_release_date(cif_path)\n\n        print(\"\\n\", target, temporal_cutoff, \"   \", template)\n        if title:\n            print(f\"PDB Title: {title}\")\n        if release_date:\n            print(f\"PDB Release Date: {release_date}\")\n\n        # sometimes there is a mismatch between PDB's fasta files and what's actually stored in coordinates,\n        # so best to get the actual residue numbers for the chain\n        chain_full_sequence, chain_sequence, chain_seq_nums = extract_rna_sequence(\n            cif_path, chain_id\n        )\n\n        # get 3d data\n        alignment = []\n        qstart = int(qstart)\n        qend = int(qend)\n        tstart = int(tstart)\n        tend = int(tend)\n        alignment.append(\n            sequence[: (qstart - 1)] + \"-\" * (tstart - 1) + qaln + sequence[qend:]\n        )\n        alignment.append(\n            \"-\" * (qstart - 1)\n            + \"X\" * (tstart - 1)\n            + taln\n            + \"-\" * (len(sequence) - qend)\n        )\n        print(alignment[0])\n        print(alignment[1])\n        c1prime_data = get_c1prime_labels(cif_path, chain_id, alignment, chain_seq_nums)\n\n        # mismatch in FASTA sequence and the polyx info in the CIF file\n        if len(c1prime_data) != len(sequence):\n            print(\n                \"WARNING! len(c1prime_data) != len(sequence)\",\n                \"len c1prime_data\",\n                len(c1prime_data),\n                \"len sequence\",\n                len(sequence),\n                \"qstart\",\n                qstart,\n                \"len qaln\",\n                len(qaln),\n                \"qend\",\n                qend,\n            )\n            continue\n\n        templates.append(c1prime_data)\n\n        if len(templates) >= MAX_TEMPLATES:\n            break\n\n    print(\"Found\", len(templates), \"templates for\", target, \"\\n\")\n\n    mapped_target = target\n    if not id_map is None:\n        mapped_target = id_map[target]\n\n    for i in range(len(sequence)):\n        output_label = {\n            \"ID\": f\"{mapped_target}_{i+1}\",\n            \"resname\": sequence[i],\n            \"resid\": i + 1,\n        }\n\n        # output templates\n        for n in range(len(templates)):\n            res, resid, x, y, z, pdb_seqnum = templates[n][i]\n            assert resid == i + 1\n            output_label[f\"x_{n+1}\"] = x\n            output_label[f\"y_{n+1}\"] = y\n            output_label[f\"z_{n+1}\"] = z\n\n        # pad with blank models\n        for n in range(len(templates), MAX_TEMPLATES):\n            output_label[f\"x_{n+1}\"] = NULL_VALUE\n            output_label[f\"y_{n+1}\"] = NULL_VALUE\n            output_label[f\"z_{n+1}\"] = NULL_VALUE\n        output_labels.append(output_label)\n\n    num_targets += 1\n    # if num_targets > 1: break # for debug!\n\n\nprint(f\"Completed {num_targets} targets\\n\")","metadata":{"_cell_guid":"ec772791-c25f-49ba-be85-19c71768c801","_kg_hide-output":true,"_uuid":"49b23e78-212e-47be-a385-cf975ad4dfe8","collapsed":false,"execution":{"iopub.execute_input":"2026-01-15T22:54:57.34219Z","iopub.status.busy":"2026-01-15T22:54:57.341832Z","iopub.status.idle":"2026-01-15T22:58:22.957171Z","shell.execute_reply":"2026-01-15T22:58:22.956093Z"},"jupyter":{"outputs_hidden":false},"papermill":{"duration":205.631127,"end_time":"2026-01-15T22:58:22.959052","exception":false,"start_time":"2026-01-15T22:54:57.327925","status":"completed"},"scrolled":true,"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Save","metadata":{"papermill":{"duration":0.014984,"end_time":"2026-01-15T22:58:22.989359","exception":false,"start_time":"2026-01-15T22:58:22.974375","status":"completed"},"tags":[]}},{"cell_type":"code","source":"output_csv(output_labels, outfile)","metadata":{"_cell_guid":"fedcf985-42ad-4bcc-b8d7-76f0acc40267","_uuid":"75d7d90c-26e8-4b22-8421-8a41ee02a73f","execution":{"iopub.execute_input":"2026-01-15T22:58:23.022623Z","iopub.status.busy":"2026-01-15T22:58:23.021944Z","iopub.status.idle":"2026-01-15T22:58:23.186182Z","shell.execute_reply":"2026-01-15T22:58:23.184992Z"},"papermill":{"duration":0.183706,"end_time":"2026-01-15T22:58:23.18813","exception":false,"start_time":"2026-01-15T22:58:23.004424","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Generate templates using last year 1st place approach\n> https://www.kaggle.com/code/jaejohn/rna-3d-folds-tbm-only-approach","metadata":{"papermill":{"duration":0.015048,"end_time":"2026-01-15T22:58:23.21825","exception":false,"start_time":"2026-01-15T22:58:23.203202","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import time\n\nimport pandas as pd\nimport numpy as np\n\nimport random\nfrom Bio import pairwise2\nfrom Bio.Seq import Seq\n\nfrom tqdm.notebook import tqdm\n\nfrom scipy.spatial.transform import Rotation as R\nfrom sklearn.preprocessing import normalize\nfrom scipy.spatial import distance_matrix\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.execute_input":"2026-01-15T22:58:23.25Z","iopub.status.busy":"2026-01-15T22:58:23.249643Z","iopub.status.idle":"2026-01-15T22:58:25.259652Z","shell.execute_reply":"2026-01-15T22:58:25.258849Z"},"papermill":{"duration":2.028483,"end_time":"2026-01-15T22:58:25.261715","exception":false,"start_time":"2026-01-15T22:58:23.233232","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Utils","metadata":{"papermill":{"duration":0.015038,"end_time":"2026-01-15T22:58:25.291919","exception":false,"start_time":"2026-01-15T22:58:25.276881","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from Bio.Seq import Seq\nfrom Bio import pairwise2\nimport numpy as np\nfrom sklearn.cluster import KMeans\nfrom sklearn.metrics.pairwise import cosine_similarity\n\nimport parasail\n\n\ndef process_labels(labels_df):\n    coords_dict = {}\n    labels_df[\"group\"] = labels_df[\"ID\"].apply(lambda x: x.rsplit(\"_\", 1)[0])\n\n    for id_prefix, group in tqdm(\n        labels_df.groupby(\"group\"), desc=\"Processing structures\"\n    ):\n        # Extract just the coordinates columns for the first structure (x_1, y_1, z_1)\n        group = group.sort_values(\"resid\")\n        coords_dict[id_prefix] = group[[\"x_1\", \"y_1\", \"z_1\"]].values\n\n    labels_df.drop(\"group\", axis=1, inplace=True)\n    return coords_dict\n\n\ndef find_similar_sequences(\n    query_seq, train_seqs_df, train_coords_dict, train_features_dict, top_n=5\n):\n    \"\"\"\n    Find similar RNA sequences using enhanced scoring and clustering for diversity.\n\n    Improvements:\n    - Multi-tier length filtering\n    - Enhanced alignment scoring with multiple algorithms\n    - RNA-specific structural features\n    - Adaptive clustering\n    \"\"\"\n    similar_seqs = []\n    query_seq_obj = Seq(query_seq)\n    query_features = _extract_enhanced_rna_features(query_seq)\n\n    # Step 1: Enhanced candidate selection with multi-tier filtering\n    for i, row in tqdm(\n        train_seqs_df.iterrows(), total=len(train_seqs_df), disable=True\n    ):\n        target_id = row[\"target_id\"]\n        train_seq = row[\"sequence\"]\n\n        # Skip if coordinates not available\n        if target_id not in train_coords_dict:\n            continue\n\n        # Multi-tier length filtering (more permissive for very short/long sequences)\n        len_ratio = abs(len(train_seq) - len(query_seq)) / max(\n            len(train_seq), len(query_seq)\n        )\n        if (\n            len(query_seq) < 50 or len(train_seq) < 50\n        ):  # Short sequences - more permissive\n            if len_ratio > 0.6:\n                continue\n        elif (\n            len(query_seq) > 1000 or len(train_seq) > 1000\n        ):  # Long sequences - stricter\n            if len_ratio > 0.2:\n                continue\n        else:  # Medium sequences - original threshold\n            if len_ratio > 0.4:\n                continue\n\n        # Calculate composite similarity score\n        composite_score = _calculate_composite_similarity(\n            query_seq, train_seq, query_features, train_features_dict[target_id]\n        )\n\n        if composite_score > 0:  # Only keep sequences with positive similarity\n            similar_seqs.append(\n                (target_id, train_seq, composite_score, train_coords_dict[target_id])\n            )\n\n    # Sort by composite score and take top candidates\n    similar_seqs.sort(key=lambda x: x[2], reverse=True)\n\n    # Adaptive candidate selection based on score distribution\n    candidate_count = min(50, len(similar_seqs))  # Increased initial pool\n    if len(similar_seqs) > 10:\n        # Filter out sequences with very low scores (bottom 20%)\n        score_threshold = np.percentile([x[2] for x in similar_seqs], 80)\n        filtered_candidates = [x for x in similar_seqs if x[2] >= score_threshold]\n        candidate_count = min(candidate_count, len(filtered_candidates))\n        top_candidates = filtered_candidates[:candidate_count]\n    else:\n        top_candidates = similar_seqs[:candidate_count]\n\n    # If we have fewer sequences than requested clusters, return all\n    if len(top_candidates) <= top_n:\n        return top_candidates[:top_n]\n\n    # Step 2: Enhanced feature matrix for better clustering\n    feature_matrix = []\n    for _, seq, _, _ in top_candidates:\n        features = _extract_enhanced_rna_features(seq)\n        feature_matrix.append(features)\n\n    feature_matrix = np.array(feature_matrix)\n\n    # Step 3: Adaptive clustering\n    n_clusters = min(top_n, len(top_candidates))\n\n    # Use different clustering approach based on dataset size\n    if len(top_candidates) >= 15:\n        # K-means for larger datasets\n        kmeans = KMeans(n_clusters=n_clusters, random_state=42, n_init=10)\n        cluster_labels = kmeans.fit_predict(feature_matrix)\n    else:\n        # Simple diversity-based selection for smaller datasets\n        cluster_labels = _diversity_based_clustering(feature_matrix, n_clusters)\n\n    # Step 4: Select best representative from each cluster\n    final_results = []\n    for cluster_id in range(n_clusters):\n        cluster_sequences = [\n            top_candidates[i]\n            for i in range(len(top_candidates))\n            if cluster_labels[i] == cluster_id\n        ]\n\n        if cluster_sequences:\n            # Sort by composite score and take the best one\n            cluster_sequences.sort(key=lambda x: x[2], reverse=True)\n            final_results.append(cluster_sequences[0])\n\n    # Sort final results by similarity score\n    final_results.sort(key=lambda x: x[2], reverse=True)\n\n    return final_results[:top_n]\n\n\nAA = \"AUCG\"\nGAP_OPEN = 100  # 10 × 10\nGAP_EXT = 5  # 0.5 × 10\nPMATCH = 29\nPMISMATCH = -10\nMATRIX = parasail.matrix_create(AA, PMATCH, PMISMATCH)\n\n\ndef fast_alignment_scores(query_seq, train_seq):\n    q = query_seq.encode()\n    t = train_seq.encode()\n    min_len = min(len(q), len(t))\n\n    # Global alignment\n    g = parasail.nw_scan_16(q, t, GAP_OPEN, GAP_EXT, MATRIX)\n    global_score = g.score / (20 * min_len)\n\n    # Local alignment\n    l = parasail.sw_scan_16(q, t, GAP_OPEN, GAP_EXT, MATRIX)\n    local_score = l.score / (20 * min_len)\n\n    return global_score, local_score\n\n\ndef _calculate_composite_similarity(\n    query_seq, train_seq, query_features, train_features\n):\n    \"\"\"\n    Calculate composite similarity using multiple alignment methods and features.\n    \"\"\"\n    query_seq_obj = Seq(query_seq)\n\n    # 1. Global alignment (original method)\n    USE_PARASAIL = True\n\n    if USE_PARASAIL:\n        global_score, local_score = fast_alignment_scores(query_seq, train_seq)\n        # print(\"PARASAIL\", global_score, local_score)\n    else:\n        global_alignments = pairwise2.align.globalms(\n            query_seq_obj, train_seq, 2.9, -1, -10, -0.5, one_alignment_only=True\n        )\n        global_score = 0\n        if global_alignments:\n            alignment = global_alignments[0]\n            global_score = alignment.score / (2 * min(len(query_seq), len(train_seq)))\n\n        # 2. Local alignment for finding similar regions\n        local_alignments = pairwise2.align.localms(\n            query_seq_obj, train_seq, 2.9, -1, -10, -0.5, one_alignment_only=True\n        )\n        local_score = 0\n        if local_alignments:\n            alignment = local_alignments[0]\n            local_score = alignment.score / (2 * min(len(query_seq), len(train_seq)))\n        # print(\"PAIRWISE\", global_score, local_score)\n\n    # 3. Feature-based similarity\n    feature_similarity = cosine_similarity([query_features], [train_features])[0][0]\n\n    # 4. K-mer similarity for sequence motifs\n    kmer_similarity = _calculate_kmer_similarity(query_seq, train_seq, k=3)\n\n    # Weighted composite score\n    composite_score = (\n        0.4 * global_score\n        + 0.3 * local_score\n        + 0.2 * feature_similarity\n        + 0.1 * kmer_similarity\n    )\n\n    return composite_score\n\n\ndef _calculate_kmer_similarity(seq1, seq2, kmers1=None, kmers2=None, k=3):\n    \"\"\"Calculate k-mer based similarity between sequences.\"\"\"\n\n    def get_kmers(seq, k):\n        return set(seq[i : i + k] for i in range(len(seq) - k + 1))\n\n    if kmers1 is None:\n        kmers1 = get_kmers(seq1.upper(), k)\n    if kmers2 is None:\n        kmers2 = get_kmers(seq2.upper(), k)\n\n    if not kmers1 or not kmers2:\n        return 0\n\n    intersection = len(kmers1.intersection(kmers2))\n    union = len(kmers1) + len(kmers2) - intersection\n    # union = len(kmers1.union(kmers2))\n\n    return intersection / union if union > 0 else 0\n\n\ndef _diversity_based_clustering(feature_matrix, n_clusters):\n    \"\"\"Simple diversity-based clustering for small datasets.\"\"\"\n    n_samples = len(feature_matrix)\n    cluster_labels = np.zeros(n_samples, dtype=int)\n\n    if n_samples <= n_clusters:\n        return np.arange(n_samples)\n\n    # Select diverse representatives\n    selected_indices = [0]  # Start with first sequence\n\n    for cluster_id in range(1, n_clusters):\n        max_min_distance = -1\n        best_idx = -1\n\n        for i in range(n_samples):\n            if i in selected_indices:\n                continue\n\n            # Find minimum distance to already selected sequences\n            min_distance = min(\n                np.linalg.norm(feature_matrix[i] - feature_matrix[j])\n                for j in selected_indices\n            )\n\n            if min_distance > max_min_distance:\n                max_min_distance = min_distance\n                best_idx = i\n\n        if best_idx != -1:\n            selected_indices.append(best_idx)\n\n    # Assign remaining sequences to closest cluster centers\n    for i in range(n_samples):\n        if i not in selected_indices:\n            distances = [\n                np.linalg.norm(feature_matrix[i] - feature_matrix[j])\n                for j in selected_indices\n            ]\n            cluster_labels[i] = np.argmin(distances)\n        else:\n            cluster_labels[i] = selected_indices.index(i)\n\n    return cluster_labels\n\n\ndef _extract_enhanced_rna_features(sequence):\n    \"\"\"\n    Extract comprehensive RNA-specific features for better clustering and similarity.\n    \"\"\"\n    seq = sequence.upper()\n    features = []\n\n    # 1. Basic nucleotide frequencies\n    nucleotides = [\"A\", \"U\", \"G\", \"C\"]\n    for nuc in nucleotides:\n        freq = seq.count(nuc) / len(seq) if len(seq) > 0 else 0\n        features.append(freq)\n\n    # 2. Dinucleotide frequencies (reduced set - most important for RNA)\n    important_dinucs = [\"AU\", \"UA\", \"GC\", \"CG\", \"GU\", \"UG\", \"AA\", \"UU\", \"GG\", \"CC\"]\n    for dinuc in important_dinucs:\n        count = 0\n        for i in range(len(seq) - 1):\n            if seq[i : i + 2] == dinuc:\n                count += 1\n        freq = count / (len(seq) - 1) if len(seq) > 1 else 0\n        features.append(freq)\n\n    # 3. RNA secondary structure indicators\n    gc_content = (seq.count(\"G\") + seq.count(\"C\")) / len(seq) if len(seq) > 0 else 0\n    au_content = (seq.count(\"A\") + seq.count(\"U\")) / len(seq) if len(seq) > 0 else 0\n    purine_content = (seq.count(\"A\") + seq.count(\"G\")) / len(seq) if len(seq) > 0 else 0\n    pyrimidine_content = (\n        (seq.count(\"U\") + seq.count(\"C\")) / len(seq) if len(seq) > 0 else 0\n    )\n\n    features.extend([gc_content, au_content, purine_content, pyrimidine_content])\n\n    # 4. Sequence complexity measures\n    length_normalized = min(len(seq) / 1000.0, 1.0)  # Capped normalization\n\n    # Simple entropy calculation\n    entropy = 0\n    for nuc in nucleotides:\n        freq = seq.count(nuc) / len(seq) if len(seq) > 0 else 0\n        if freq > 0:\n            entropy -= freq * np.log2(freq)\n    entropy_normalized = entropy / 2.0  # Max entropy for 4 nucleotides is 2\n\n    features.extend([length_normalized, entropy_normalized])\n\n    # 5. Repetitive pattern detection\n    repeat_content = _calculate_repeat_content(seq)\n    features.append(repeat_content)\n\n    return features\n\n\ndef _calculate_repeat_content(sequence):\n    \"\"\"Calculate the proportion of repetitive content in the sequence.\"\"\"\n    if len(sequence) < 6:\n        return 0\n\n    repeat_count = 0\n    window_size = 3\n\n    for i in range(len(sequence) - window_size + 1):\n        motif = sequence[i : i + window_size]\n        # Look for the same motif in the rest of the sequence\n        for j in range(i + window_size, len(sequence) - window_size + 1):\n            if sequence[j : j + window_size] == motif:\n                repeat_count += 1\n                break\n\n    return (\n        repeat_count / (len(sequence) - window_size + 1)\n        if len(sequence) > window_size\n        else 0\n    )\n\n\ndef adaptive_rna_constraints(coordinates, sequence, confidence=1.0):\n    # Make a copy of coordinates to refine\n    refined_coords = coordinates.copy()\n    n_residues = len(sequence)\n\n    # Calculate constraint strength (inverse of confidence)\n    # High confidence templates receive gentler constraints\n    constraint_strength = 0.8 * (1.0 - min(confidence, 0.8))\n\n    # 1. Sequential distance constraints (consecutive nucleotides)\n    # More flexible distance range (statistical distribution from PDB)\n    seq_min_dist = 5.5  # Minimum sequential distance\n    seq_max_dist = 6.5  # Maximum sequential distance\n\n    for i in range(n_residues - 1):\n        current_pos = refined_coords[i]\n        next_pos = refined_coords[i + 1]\n\n        # Calculate current distance\n        current_dist = np.linalg.norm(next_pos - current_pos)\n\n        # Only adjust if significantly outside expected range\n        if current_dist < seq_min_dist or current_dist > seq_max_dist:\n            # Calculate target distance (midpoint of range)\n            target_dist = (seq_min_dist + seq_max_dist) / 2\n\n            # Get direction vector\n            direction = next_pos - current_pos\n            direction = direction / (np.linalg.norm(direction) + 1e-10)\n\n            # Apply partial adjustment based on constraint strength\n            adjustment = (target_dist - current_dist) * constraint_strength\n\n            # Only adjust the next position to preserve the overall fold\n            refined_coords[i + 1] = current_pos + direction * (\n                current_dist + adjustment\n            )\n\n    # 2. Steric clash prevention (more conservative)\n    min_allowed_distance = 3.8  # Minimum distance between non-consecutive C1' atoms\n\n    # Calculate all pairwise distances\n    dist_matrix = distance_matrix(refined_coords, refined_coords)\n\n    # Find severe clashes (atoms too close)\n    severe_clashes = np.where((dist_matrix < min_allowed_distance) & (dist_matrix > 0))\n\n    # Fix severe clashes\n    for idx in range(len(severe_clashes[0])):\n        i, j = severe_clashes[0][idx], severe_clashes[1][idx]\n\n        # Skip consecutive nucleotides and previously processed pairs\n        if abs(i - j) <= 1 or i >= j:\n            continue\n\n        # Get current positions and distance\n        pos_i = refined_coords[i]\n        pos_j = refined_coords[j]\n        current_dist = dist_matrix[i, j]\n\n        # Calculate necessary adjustment but scale by constraint strength\n        direction = pos_j - pos_i\n        direction = direction / (np.linalg.norm(direction) + 1e-10)\n\n        # Calculate partial adjustment\n        adjustment = (min_allowed_distance - current_dist) * constraint_strength\n\n        # Move points apart\n        refined_coords[i] = pos_i - direction * (adjustment / 2)\n        refined_coords[j] = pos_j + direction * (adjustment / 2)\n\n    # 3. Very light base-pair constraining (if confidence is low)\n    if constraint_strength > 0.3:  # Only apply if template confidence is low\n        # Simple Watson-Crick base pairs\n        pairs = {\"A\": \"U\", \"U\": \"A\", \"G\": \"C\", \"C\": \"G\"}\n\n        # Scan for potential base pairs\n        for i in range(n_residues):\n            base_i = sequence[i]\n            complement = pairs.get(base_i)\n\n            if not complement:\n                continue\n\n            # Look for complementary bases within a reasonable range\n            for j in range(i + 3, min(i + 20, n_residues)):\n                if sequence[j] == complement:\n                    # Calculate current distance\n                    current_dist = np.linalg.norm(refined_coords[i] - refined_coords[j])\n\n                    # Only consider if distance suggests potential pairing\n                    if 8.0 < current_dist < 14.0:\n                        # Target 10.5Å as generic base-pair C1'-C1' distance\n                        target_dist = 10.5\n\n                        # Calculate very gentle adjustment (scaled by constraint_strength)\n                        adjustment = (target_dist - current_dist) * (\n                            constraint_strength * 0.3\n                        )\n\n                        # Get direction vector\n                        direction = refined_coords[j] - refined_coords[i]\n                        direction = direction / (np.linalg.norm(direction) + 1e-10)\n\n                        # Apply very gentle adjustment to both positions\n                        refined_coords[i] = refined_coords[i] - direction * (\n                            adjustment / 2\n                        )\n                        refined_coords[j] = refined_coords[j] + direction * (\n                            adjustment / 2\n                        )\n\n                        # Only consider one potential pair per base (closest match)\n                        break\n\n    return refined_coords\n\n\ndef adapt_template_to_query(query_seq, template_seq, template_coords, alignment=None):\n    if alignment is None:\n        from Bio.Seq import Seq\n        from Bio import pairwise2\n\n        query_seq_obj = Seq(query_seq)\n        template_seq_obj = Seq(template_seq)\n        alignments = pairwise2.align.globalms(\n            query_seq_obj, template_seq_obj, 2.9, -1, -10, -0.5, one_alignment_only=True\n        )\n\n        if not alignments:\n            return generate_improved_rna_structure(query_seq)\n\n        alignment = alignments[0]\n\n    aligned_query = alignment.seqA\n    aligned_template = alignment.seqB\n\n    query_coords = np.zeros((len(query_seq), 3))\n    query_coords.fill(np.nan)\n\n    # Map template coordinates to query\n    query_idx = 0\n    template_idx = 0\n\n    for i in range(len(aligned_query)):\n        query_char = aligned_query[i]\n        template_char = aligned_template[i]\n\n        if query_char != \"-\" and template_char != \"-\":\n            if template_idx < len(template_coords):\n                query_coords[query_idx] = template_coords[template_idx]\n            template_idx += 1\n            query_idx += 1\n        elif query_char != \"-\" and template_char == \"-\":\n            query_idx += 1\n        elif query_char == \"-\" and template_char != \"-\":\n            template_idx += 1\n\n    # IMPROVED GAP FILLING - maintains RNA backbone geometry\n    backbone_distance = 5.9  # Typical C1'-C1' distance\n\n    # Fill gaps by maintaining realistic backbone connectivity\n    for i in range(len(query_coords)):\n        if np.isnan(query_coords[i, 0]):\n            # Find nearest valid neighbors\n            prev_valid = next_valid = None\n\n            for j in range(i - 1, -1, -1):\n                if not np.isnan(query_coords[j, 0]):\n                    prev_valid = j\n                    break\n\n            for j in range(i + 1, len(query_coords)):\n                if not np.isnan(query_coords[j, 0]):\n                    next_valid = j\n                    break\n\n            if prev_valid is not None and next_valid is not None:\n                # Interpolate along realistic RNA backbone path\n                gap_size = next_valid - prev_valid\n                total_distance = np.linalg.norm(\n                    query_coords[next_valid] - query_coords[prev_valid]\n                )\n                expected_distance = gap_size * backbone_distance\n\n                # If gap is compressed, extend it realistically\n                if total_distance < expected_distance * 0.7:\n                    direction = query_coords[next_valid] - query_coords[prev_valid]\n                    direction = direction / (np.linalg.norm(direction) + 1e-10)\n\n                    # Place intermediate points along extended path\n                    for k, idx in enumerate(range(prev_valid + 1, next_valid)):\n                        progress = (k + 1) / gap_size\n                        base_pos = (\n                            query_coords[prev_valid]\n                            + direction * expected_distance * progress\n                        )\n\n                        # Add slight curvature for realism\n                        perpendicular = np.cross(direction, [0, 0, 1])\n                        if np.linalg.norm(perpendicular) < 1e-6:\n                            perpendicular = np.cross(direction, [1, 0, 0])\n                        perpendicular = perpendicular / (\n                            np.linalg.norm(perpendicular) + 1e-10\n                        )\n\n                        curve_amplitude = 2.0 * np.sin(progress * np.pi)\n                        query_coords[idx] = base_pos + perpendicular * curve_amplitude\n                else:\n                    # Linear interpolation for normal gaps\n                    for k, idx in enumerate(range(prev_valid + 1, next_valid)):\n                        weight = (k + 1) / gap_size\n                        query_coords[idx] = (1 - weight) * query_coords[\n                            prev_valid\n                        ] + weight * query_coords[next_valid]\n\n            elif prev_valid is not None:\n                # Extend from previous position\n                if prev_valid > 0 and not np.isnan(query_coords[prev_valid - 1, 0]):\n                    direction = query_coords[prev_valid] - query_coords[prev_valid - 1]\n                    direction = direction / (np.linalg.norm(direction) + 1e-10)\n                else:\n                    direction = np.array([1.0, 0.0, 0.0])\n\n                steps_needed = i - prev_valid\n                for step in range(1, steps_needed + 1):\n                    pos_idx = prev_valid + step\n                    if pos_idx < len(query_coords):\n                        query_coords[pos_idx] = (\n                            query_coords[prev_valid]\n                            + direction * backbone_distance * step\n                        )\n\n            elif next_valid is not None:\n                # Work backwards from next position\n                direction = np.array([-1.0, 0.0, 0.0])  # Default backward direction\n                steps_needed = next_valid - i\n                for step in range(steps_needed, 0, -1):\n                    pos_idx = next_valid - step\n                    if pos_idx >= 0:\n                        query_coords[pos_idx] = (\n                            query_coords[next_valid]\n                            - direction * backbone_distance * step\n                        )\n\n    # Final cleanup\n    query_coords = np.nan_to_num(query_coords)\n    return query_coords\n\n\ndef generate_improved_rna_structure(sequence):\n    \"\"\"\n    Generate a more realistic RNA structure fallback based on sequence patterns\n    and basic RNA structure principles.\n\n    Args:\n        sequence: RNA sequence string\n\n    Returns:\n        Array of 3D coordinates\n    \"\"\"\n    n_residues = len(sequence)\n    coordinates = np.zeros((n_residues, 3))\n\n    # Analyze sequence to predict structural elements\n    # Look for complementary regions that could form base pairs\n    potential_stems = identify_potential_stems(sequence)\n\n    # Default parameters\n    radius_helix = 10.0\n    radius_loop = 15.0\n    rise_per_residue_helix = 2.5\n    rise_per_residue_loop = 1.5\n    angle_per_residue_helix = 0.6\n    angle_per_residue_loop = 0.3\n\n    # Assign structural classifications\n    structure_types = assign_structure_types(sequence, potential_stems)\n\n    # Generate coordinates based on predicted structure\n    current_pos = np.array([0.0, 0.0, 0.0])\n    current_direction = np.array([0.0, 0.0, 1.0])\n    current_angle = 0.0\n\n    for i in range(n_residues):\n        if structure_types[i] == \"stem\":\n            # Part of a helical stem\n            current_angle += angle_per_residue_helix\n            coordinates[i] = [\n                radius_helix * np.cos(current_angle),\n                radius_helix * np.sin(current_angle),\n                current_pos[2] + rise_per_residue_helix,\n            ]\n            current_pos = coordinates[i]\n        elif structure_types[i] == \"loop\":\n            # Part of a loop\n            current_angle += angle_per_residue_loop\n            z_shift = rise_per_residue_loop * np.sin(current_angle * 0.5)\n            coordinates[i] = [\n                radius_loop * np.cos(current_angle),\n                radius_loop * np.sin(current_angle),\n                current_pos[2] + z_shift,\n            ]\n            current_pos = coordinates[i]\n        else:\n            # Single-stranded region\n            # Add some randomness to make it look more realistic\n            jitter = np.random.normal(0, 1, 3) * 2.0\n            coordinates[i] = current_pos + jitter\n            current_pos = coordinates[i]\n\n    return coordinates\n\n\ndef identify_potential_stems(sequence):\n    \"\"\"\n    Identify potential stem regions by looking for self-complementary segments.\n\n    Args:\n        sequence: RNA sequence string\n\n    Returns:\n        List of tuples (start1, end1, start2, end2) representing potentially paired regions\n    \"\"\"\n    complementary_bases = {\"A\": \"U\", \"U\": \"A\", \"G\": \"C\", \"C\": \"G\"}\n    min_stem_length = 3\n    potential_stems = []\n\n    # Simple stem identification\n    for i in range(len(sequence) - min_stem_length):\n        for j in range(i + min_stem_length + 3, len(sequence) - min_stem_length + 1):\n            # Check if regions could form a stem\n            potential_stem_len = min(min_stem_length, len(sequence) - j)\n            is_stem = True\n\n            for k in range(potential_stem_len):\n                if (\n                    sequence[i + k] not in complementary_bases\n                    or complementary_bases[sequence[i + k]]\n                    != sequence[j + potential_stem_len - k - 1]\n                ):\n                    is_stem = False\n                    break\n\n            if is_stem:\n                potential_stems.append(\n                    (i, i + potential_stem_len - 1, j, j + potential_stem_len - 1)\n                )\n\n    return potential_stems\n\n\ndef assign_structure_types(sequence, potential_stems):\n    \"\"\"\n    Assign each nucleotide to a structural element type.\n\n    Args:\n        sequence: RNA sequence string\n        potential_stems: List of tuples representing stem regions\n\n    Returns:\n        List of structure types ('stem', 'loop', 'single')\n    \"\"\"\n    structure_types = [\"single\"] * len(sequence)\n\n    # Mark stem regions\n    for stem in potential_stems:\n        start1, end1, start2, end2 = stem\n        for i in range(end1 - start1 + 1):\n            structure_types[start1 + i] = \"stem\"\n            structure_types[end2 - i] = \"stem\"\n\n    # Mark loop regions (regions between paired regions)\n    for i in range(len(potential_stems) - 1):\n        _, end1, start2, _ = potential_stems[i]\n        next_start1, _, _, _ = potential_stems[i + 1]\n\n        if next_start1 > end1 + 1 and start2 > next_start1:\n            for j in range(end1 + 1, next_start1):\n                structure_types[j] = \"loop\"\n\n    return structure_types\n\n\n# Function to create a more realistic RNA structure when no good templates are found\ndef generate_rna_structure(sequence, seed=None):\n    if seed is not None:\n        np.random.seed(seed)\n        random.seed(seed)\n\n    n_residues = len(sequence)\n    coordinates = np.zeros((n_residues, 3))\n\n    # Initialize the first few residues in a helix\n    for i in range(min(3, n_residues)):\n        angle = i * 0.6\n        coordinates[i] = [10.0 * np.cos(angle), 10.0 * np.sin(angle), i * 2.5]\n\n    # Add more complex folding patterns\n    current_direction = np.array([0.0, 0.0, 1.0])  # Start moving along z-axis\n\n    # Define base-pairing tendencies (G-C and A-U pairs)\n    for i in range(3, n_residues):\n        # Check for potential base-pairing in the sequence\n        has_pair = False\n        pair_idx = -1\n\n        # Simple detection of complementary bases (G-C, A-U)\n        complementary = {\"G\": \"C\", \"C\": \"G\", \"A\": \"U\", \"U\": \"A\"}\n        current_base = sequence[i]\n\n        # Look for potential base-pairing within a window before the current position\n        window_size = min(i, 15)  # Look back up to 15 bases\n        for j in range(i - window_size, i):\n            if j >= 0 and sequence[j] == complementary.get(current_base, \"X\"):\n                # Found a potential pair\n                has_pair = True\n                pair_idx = j\n                break\n\n        if has_pair and i - pair_idx <= 10 and random.random() < 0.7:\n            # Try to create a base-pair by positioning this nucleotide near its pair\n            pair_pos = coordinates[pair_idx]\n\n            # Create a position that's roughly opposite to the pair\n            random_offset = np.random.normal(0, 1, 3) * 2.0\n            base_pair_distance = 10.0 + random.uniform(-1.0, 1.0)\n\n            # Calculate a vector from base-pair toward center of structure\n            center = np.mean(coordinates[:i], axis=0)\n            direction = center - pair_pos\n            direction = direction / (np.linalg.norm(direction) + 1e-10)\n\n            # Position new nucleotide in the general direction of the \"center\"\n            coordinates[i] = pair_pos + direction * base_pair_distance + random_offset\n\n            # Update direction for next nucleotide\n            current_direction = np.random.normal(0, 0.3, 3)\n            current_direction = current_direction / (\n                np.linalg.norm(current_direction) + 1e-10\n            )\n\n        else:\n            # No base-pairing detected, continue with the current fold direction\n            # Randomly rotate current direction to simulate RNA flexibility\n            if random.random() < 0.3:\n                # More significant direction change\n                angle = random.uniform(0.2, 0.6)\n                axis = np.random.normal(0, 1, 3)\n                axis = axis / (np.linalg.norm(axis) + 1e-10)\n                rotation = R.from_rotvec(angle * axis)\n                current_direction = rotation.apply(current_direction)\n            else:\n                # Small random changes in direction\n                current_direction += np.random.normal(0, 0.15, 3)\n                current_direction = current_direction / (\n                    np.linalg.norm(current_direction) + 1e-10\n                )\n\n            # Distance between consecutive nucleotides (3.5-4.5Å is typical)\n            step_size = random.uniform(3.5, 4.5)\n\n            # Update position\n            coordinates[i] = coordinates[i - 1] + step_size * current_direction\n\n    return coordinates\n\n\ndef predict_rna_structures(\n    sequence,\n    target_id,\n    train_seqs_df,\n    train_coords_dict,\n    train_features_dict,\n    n_predictions=5,\n):\n    predictions = []\n\n    # Find similar sequences in the training data\n    similar_seqs = find_similar_sequences(\n        sequence,\n        train_seqs_df,\n        train_coords_dict,\n        train_features_dict,\n        top_n=n_predictions,\n    )\n\n    # If we found any similar sequences, use them as templates\n    if similar_seqs:\n        for i, (\n            template_id,\n            template_seq,\n            similarity_score,\n            template_coords,\n        ) in enumerate(similar_seqs):\n            # Adapt template coordinates to the query sequence\n            adapted_coords = adapt_template_to_query(\n                sequence, template_seq, template_coords\n            )\n\n            if adapted_coords is not None:\n                # Apply adaptive constraints based on template similarity\n                # For high similarity templates, apply very gentle constraints\n                refined_coords = adaptive_rna_constraints(\n                    adapted_coords, sequence, confidence=similarity_score\n                )\n\n                # Add some randomness (less for better templates)\n                random_scale = max(0.05, 0.8 - similarity_score)  # Reduced randomness\n                randomized_coords = refined_coords.copy()\n                randomized_coords += np.random.normal(\n                    0, random_scale, randomized_coords.shape\n                )\n\n                predictions.append(randomized_coords)\n\n                if len(predictions) >= n_predictions:\n                    break\n\n    # If we don't have enough predictions from templates, generate de novo structures\n    while len(predictions) < n_predictions:\n        seed_value = hash(target_id) % 10000 + len(predictions) * 1000\n        de_novo_coords = generate_rna_structure(sequence, seed=seed_value)\n\n        # Apply stronger constraints to de novo structures (lower confidence)\n        refined_de_novo = adaptive_rna_constraints(\n            de_novo_coords, sequence, confidence=0.2\n        )\n\n        predictions.append(refined_de_novo)\n\n    return predictions[:n_predictions]\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2026-01-15T22:58:25.324982Z","iopub.status.busy":"2026-01-15T22:58:25.324294Z","iopub.status.idle":"2026-01-15T22:58:26.095685Z","shell.execute_reply":"2026-01-15T22:58:26.094714Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.791069,"end_time":"2026-01-15T22:58:26.09792","exception":false,"start_time":"2026-01-15T22:58:25.306851","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Main","metadata":{"papermill":{"duration":0.015271,"end_time":"2026-01-15T22:58:26.128336","exception":false,"start_time":"2026-01-15T22:58:26.113065","status":"completed"},"tags":[]}},{"cell_type":"code","source":"COMP = \"/kaggle/input/stanford-rna-3d-folding-2/\"\n\ntrain_seqs = pd.read_csv(COMP + '/train_sequences.csv')\nvalid_seqs = pd.read_csv(COMP + '/validation_sequences.csv')\ntest_seqs = pd.read_csv(COMP + '/test_sequences.csv')\ntrain_labels = pd.read_csv(COMP + '/train_labels.csv')\nvalid_labels = pd.read_csv(COMP + '/validation_labels.csv')\n\nprint(f\"Loaded {len(train_seqs)} training sequences, {len(valid_seqs)} validation sequences, and {len(test_seqs)} test sequences\")","metadata":{"execution":{"iopub.execute_input":"2026-01-15T22:58:26.161283Z","iopub.status.busy":"2026-01-15T22:58:26.160187Z","iopub.status.idle":"2026-01-15T22:58:36.964878Z","shell.execute_reply":"2026-01-15T22:58:36.963924Z"},"papermill":{"duration":10.823296,"end_time":"2026-01-15T22:58:36.966863","exception":false,"start_time":"2026-01-15T22:58:26.143567","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_coords_dict = process_labels(train_labels)","metadata":{"execution":{"iopub.execute_input":"2026-01-15T22:58:36.99906Z","iopub.status.busy":"2026-01-15T22:58:36.998684Z","iopub.status.idle":"2026-01-15T22:58:47.461283Z","shell.execute_reply":"2026-01-15T22:58:47.460217Z"},"papermill":{"duration":10.481265,"end_time":"2026-01-15T22:58:47.46341","exception":false,"start_time":"2026-01-15T22:58:36.982145","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_features = [_extract_enhanced_rna_features(train_seq) for train_seq in tqdm(train_seqs['sequence'])]","metadata":{"execution":{"iopub.execute_input":"2026-01-15T22:58:47.496266Z","iopub.status.busy":"2026-01-15T22:58:47.495909Z","iopub.status.idle":"2026-01-15T22:59:59.136614Z","shell.execute_reply":"2026-01-15T22:59:59.135701Z"},"papermill":{"duration":71.659187,"end_time":"2026-01-15T22:59:59.138397","exception":false,"start_time":"2026-01-15T22:58:47.47921","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_features_dict = dict(zip(train_seqs['target_id'], train_features))","metadata":{"execution":{"iopub.execute_input":"2026-01-15T22:59:59.171811Z","iopub.status.busy":"2026-01-15T22:59:59.170865Z","iopub.status.idle":"2026-01-15T22:59:59.176925Z","shell.execute_reply":"2026-01-15T22:59:59.176092Z"},"papermill":{"duration":0.024729,"end_time":"2026-01-15T22:59:59.178698","exception":false,"start_time":"2026-01-15T22:59:59.153969","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List to store all prediction records\nall_predictions = []\n\n# Set up time tracking\nstart_time = time.time()\ntotal_targets = len(test_seqs)\n\n# For each sequence in the test set\nfor idx, row in tqdm(test_seqs.iterrows(), total=len(test_seqs)):\n    target_id = row[\"target_id\"]\n    sequence = row[\"sequence\"]\n\n    # Generate 5 different structure predictions\n    predictions = predict_rna_structures(\n        sequence,\n        target_id,\n        train_seqs,\n        train_coords_dict,\n        train_features_dict,\n        n_predictions=5,\n    )\n\n    # For each residue in the sequence\n    for j in range(len(sequence)):\n        pred_row = {\"ID\": f\"{target_id}_{j+1}\", \"resname\": sequence[j], \"resid\": j + 1}\n\n        # Add coordinates from all 5 predictions\n        for i in range(5):\n            pred_row[f\"x_{i+1}\"] = predictions[i][j][0]\n            pred_row[f\"y_{i+1}\"] = predictions[i][j][1]\n            pred_row[f\"z_{i+1}\"] = predictions[i][j][2]\n\n        all_predictions.append(pred_row)\n\n# Create DataFrame with predictions\nsubmission_df = pd.DataFrame(all_predictions)\n\n# Ensure the submission file has the correct format\ncolumn_order = [\"ID\", \"resname\", \"resid\"]\nfor i in range(1, 6):\n    for coord in [\"x\", \"y\", \"z\"]:\n        column_order.append(f\"{coord}_{i}\")\nsubmission_df = submission_df[column_order]\n\n# Save the submission file\nsubmission_df.to_csv(\"templates_tbm.csv\", index=False)\nprint(f\"Saved predictions for {len(test_seqs)} RNA sequences to templates_tbm.csv\")","metadata":{"execution":{"iopub.execute_input":"2026-01-15T22:59:59.211224Z","iopub.status.busy":"2026-01-15T22:59:59.210872Z","iopub.status.idle":"2026-01-15T23:01:18.27366Z","shell.execute_reply":"2026-01-15T23:01:18.272635Z"},"papermill":{"duration":79.081754,"end_time":"2026-01-15T23:01:18.275869","exception":false,"start_time":"2026-01-15T22:59:59.194115","status":"completed"},"scrolled":true,"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Clean output","metadata":{"papermill":{"duration":0.01568,"end_time":"2026-01-15T23:01:18.307316","exception":false,"start_time":"2026-01-15T23:01:18.291636","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!rm -r mmseqs pdb_* *.txt *fasta tmp\n!ls","metadata":{"execution":{"iopub.execute_input":"2026-01-15T23:01:18.342191Z","iopub.status.busy":"2026-01-15T23:01:18.341415Z","iopub.status.idle":"2026-01-15T23:01:18.695172Z","shell.execute_reply":"2026-01-15T23:01:18.693955Z"},"papermill":{"duration":0.374564,"end_time":"2026-01-15T23:01:18.697545","exception":false,"start_time":"2026-01-15T23:01:18.322981","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Done !","metadata":{"papermill":{"duration":0.016331,"end_time":"2026-01-15T23:01:18.730041","exception":false,"start_time":"2026-01-15T23:01:18.71371","status":"completed"},"tags":[]}}]}