{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":15231210,"isSourceIdPinned":false},{"sourceType":"datasetVersion","sourceId":14604295,"datasetId":9328538,"databundleVersionId":15440074},{"sourceType":"datasetVersion","sourceId":11695366,"datasetId":6791615,"databundleVersionId":12170579},{"sourceType":"datasetVersion","sourceId":5123458,"datasetId":2975803,"databundleVersionId":5194879},{"sourceType":"datasetVersion","sourceId":14557731,"datasetId":9298331,"databundleVersionId":15389233},{"sourceType":"datasetVersion","sourceId":10923077,"datasetId":6785143,"databundleVersionId":11294302},{"sourceType":"datasetVersion","sourceId":14636099,"datasetId":9349494,"databundleVersionId":15474901},{"sourceType":"datasetVersion","sourceId":14519720,"datasetId":9271415,"databundleVersionId":15347344},{"sourceType":"datasetVersion","sourceId":14691128,"datasetId":9379405,"databundleVersionId":15534948},{"sourceType":"datasetVersion","sourceId":10855324,"datasetId":6742586,"databundleVersionId":11219268},{"sourceType":"modelInstanceVersion","sourceId":311741,"databundleVersionId":11641144,"modelInstanceId":264400,"modelId":285488,"isSourceIdPinned":false}],"dockerImageVersionId":31260,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":2682.412692,"end_time":"2026-02-16T20:01:05.252858","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-02-16T19:16:22.840166","version":"2.6.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"033e9082c9944cf4b4197d86bb727329":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"085c2d0997904a0882e85be38369cc2b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_7ebeca875e2647d7a7c1efcabd452172","IPY_MODEL_b8c245b2e0fd4951861ff2455cf0caef","IPY_MODEL_7f0b4d13b6d14e228c950bfb04407dcf"],"layout":"IPY_MODEL_d3e8257dd26b41a8bd93ccd215898934","tabbable":null,"tooltip":null}},"2ac82b8a25684bb38528a653fa73d16d":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_a7b3cb46b5d44d31bfb9a94db94f7179","placeholder":"​","style":"IPY_MODEL_fe011d15b7f44c878653b0f365ebb0dc","tabbable":null,"tooltip":null,"value":" 28/28 [00:00&lt;00:00, 525.02it/s]"}},"59d5f176bc53441d9a890203cb131fea":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"63f1a8847cd2465eb3198dcaecfa23d1":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_64e56c845f2f4f7089b64520358d8364","placeholder":"​","style":"IPY_MODEL_d8abb32565c24a228dc507b5bdc41e7e","tabbable":null,"tooltip":null,"value":"100%"}},"64e56c845f2f4f7089b64520358d8364":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"6a89610db4094901ac7d8a6d357a60fe":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"7459fe02dfd24940a9bdb8b1595e7d0a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"7ebeca875e2647d7a7c1efcabd452172":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_59d5f176bc53441d9a890203cb131fea","placeholder":"​","style":"IPY_MODEL_6a89610db4094901ac7d8a6d357a60fe","tabbable":null,"tooltip":null,"value":"100%"}},"7f0b4d13b6d14e228c950bfb04407dcf":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_90a3a9ebbaec4c62830c81de8c31ce8a","placeholder":"​","style":"IPY_MODEL_9921122f71254724bb67227a09870ed9","tabbable":null,"tooltip":null,"value":" 5744/5744 [00:10&lt;00:00, 689.61it/s]"}},"834685e013fb4a3cbb636ca20368a804":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8d3badb0e5254f83a6031460c9582964":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_033e9082c9944cf4b4197d86bb727329","max":28,"min":0,"orientation":"horizontal","style":"IPY_MODEL_a58fb971fcee4328aa19b944e330359b","tabbable":null,"tooltip":null,"value":28}},"90a3a9ebbaec4c62830c81de8c31ce8a":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"9921122f71254724bb67227a09870ed9":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"a58fb971fcee4328aa19b944e330359b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"a7b3cb46b5d44d31bfb9a94db94f7179":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"b8c245b2e0fd4951861ff2455cf0caef":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_834685e013fb4a3cbb636ca20368a804","max":5744,"min":0,"orientation":"horizontal","style":"IPY_MODEL_7459fe02dfd24940a9bdb8b1595e7d0a","tabbable":null,"tooltip":null,"value":5744}},"d3e8257dd26b41a8bd93ccd215898934":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"d8abb32565c24a228dc507b5bdc41e7e":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"eeb37e3521bc4a71be40b03ca42f7d5d":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"f021b7ab087b48289601f2debc6c489b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_63f1a8847cd2465eb3198dcaecfa23d1","IPY_MODEL_8d3badb0e5254f83a6031460c9582964","IPY_MODEL_2ac82b8a25684bb38528a653fa73d16d"],"layout":"IPY_MODEL_eeb37e3521bc4a71be40b03ca42f7d5d","tabbable":null,"tooltip":null}},"fe011d15b7f44c878653b0f365ebb0dc":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"8feb2cd6","cell_type":"markdown","source":"Reference\n\n- https://www.kaggle.com/code/theoviel/stanford-rna-3d-folding-pt2-rnapro-inference","metadata":{"papermill":{"duration":0.010065,"end_time":"2026-02-16T19:16:25.328871","exception":false,"start_time":"2026-02-16T19:16:25.318806","status":"completed"},"tags":[]}},{"id":"ab0509e9","cell_type":"code","source":"!pip install -q --no-index /kaggle/input/biopython-cp312/biopython-1.86-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:16:25.347379Z","iopub.status.busy":"2026-02-16T19:16:25.347112Z","iopub.status.idle":"2026-02-16T19:16:28.442546Z","shell.execute_reply":"2026-02-16T19:16:28.441719Z"},"papermill":{"duration":3.106841,"end_time":"2026-02-16T19:16:28.444442","exception":false,"start_time":"2026-02-16T19:16:25.337601","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"97397351","cell_type":"code","source":"import os\nIS_SCORING_RUN = os.environ.get('KAGGLE_IS_COMPETITION_RERUN')\nprint(IS_SCORING_RUN)","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:16:28.463597Z","iopub.status.busy":"2026-02-16T19:16:28.463342Z","iopub.status.idle":"2026-02-16T19:16:28.467761Z","shell.execute_reply":"2026-02-16T19:16:28.467122Z"},"papermill":{"duration":0.015818,"end_time":"2026-02-16T19:16:28.46934","exception":false,"start_time":"2026-02-16T19:16:28.453522","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"49407496","cell_type":"code","source":"!cp -r /kaggle/input/rnapro-src/RNAPro .\n# !cp /kaggle/input/rnapro-src/rnapro-private-best-500m.ckpt .\n# !cp /kaggle/input/rnapro-bf16-weights/5999.pt .\n# !cp /kaggle/input/rnapro-bf16-weights/5999_ema_0.995.pt .","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:16:28.488488Z","iopub.status.busy":"2026-02-16T19:16:28.488063Z","iopub.status.idle":"2026-02-16T19:16:29.281965Z","shell.execute_reply":"2026-02-16T19:16:29.281131Z"},"papermill":{"duration":0.805659,"end_time":"2026-02-16T19:16:29.283619","exception":false,"start_time":"2026-02-16T19:16:28.47796","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"d2bc22ac","cell_type":"code","source":"cd RNAPro","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:16:29.303392Z","iopub.status.busy":"2026-02-16T19:16:29.302725Z","iopub.status.idle":"2026-02-16T19:16:29.308734Z","shell.execute_reply":"2026-02-16T19:16:29.307985Z"},"papermill":{"duration":0.017358,"end_time":"2026-02-16T19:16:29.310194","exception":false,"start_time":"2026-02-16T19:16:29.292836","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"c61f44b2","cell_type":"code","source":"pwd","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:16:29.329344Z","iopub.status.busy":"2026-02-16T19:16:29.329102Z","iopub.status.idle":"2026-02-16T19:16:29.334323Z","shell.execute_reply":"2026-02-16T19:16:29.333717Z"},"papermill":{"duration":0.016375,"end_time":"2026-02-16T19:16:29.335568","exception":false,"start_time":"2026-02-16T19:16:29.319193","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"3264fe3c","cell_type":"code","source":"pip install -e . --no-deps","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:16:29.354585Z","iopub.status.busy":"2026-02-16T19:16:29.354316Z","iopub.status.idle":"2026-02-16T19:16:34.013865Z","shell.execute_reply":"2026-02-16T19:16:34.012904Z"},"papermill":{"duration":4.670959,"end_time":"2026-02-16T19:16:34.015446","exception":false,"start_time":"2026-02-16T19:16:29.344487","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"12aea038","cell_type":"code","source":"cd ..","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:16:34.034958Z","iopub.status.busy":"2026-02-16T19:16:34.034673Z","iopub.status.idle":"2026-02-16T19:16:34.039678Z","shell.execute_reply":"2026-02-16T19:16:34.038987Z"},"papermill":{"duration":0.016181,"end_time":"2026-02-16T19:16:34.041068","exception":false,"start_time":"2026-02-16T19:16:34.024887","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"f46ef08a","cell_type":"code","source":"pwd","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:16:34.060105Z","iopub.status.busy":"2026-02-16T19:16:34.059478Z","iopub.status.idle":"2026-02-16T19:16:34.063775Z","shell.execute_reply":"2026-02-16T19:16:34.06327Z"},"papermill":{"duration":0.015171,"end_time":"2026-02-16T19:16:34.065109","exception":false,"start_time":"2026-02-16T19:16:34.049938","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"21212158","cell_type":"markdown","source":"TBM\n---\n","metadata":{"papermill":{"duration":0.008792,"end_time":"2026-02-16T19:16:34.082762","exception":false,"start_time":"2026-02-16T19:16:34.07397","status":"completed"},"tags":[]}},{"id":"73c81b25","cell_type":"code","source":"%%writefile gen_submission_tbm.py\n\n# Cell 1: Imports and Setup\nimport os\nimport pandas as pd\nimport numpy as np\nfrom scipy.spatial.transform import Rotation as R\nfrom scipy.spatial import distance_matrix\nimport random\nimport hashlib\nimport time\nimport warnings\nfrom joblib import Parallel, delayed\n\nfrom tqdm.auto import tqdm\nwarnings.filterwarnings('ignore')\nfrom Bio.Align import PairwiseAligner\nfrom Bio.Seq import Seq\n\nIS_SCORING_RUN = os.environ.get('KAGGLE_IS_COMPETITION_RERUN')\n\n# Enable tqdm for pandas operations\ntqdm.pandas()\n    \n# Initialize global aligner with RNA-appropriate parameters\nALIGNER = PairwiseAligner()\nALIGNER.mode = 'global'\nALIGNER.match_score = 2.5\nALIGNER.mismatch_score = -1\nALIGNER.open_gap_score = -10\nALIGNER.extend_gap_score = -0.5\n\nseed = 21\nnp.random.seed(seed)\nrandom.seed(seed)\n\n# Cell 2: Load Data\nBASE_PATH = '/kaggle/input/stanford-rna-3d-folding-2'\n\n# Load with progress indication\ntest_seqs = pd.read_csv(f'{BASE_PATH}/test_sequences.csv')\ntrain_seqs = pd.read_csv('/kaggle/input/stanford-rna-3d-folding-pdb-parsed-data/all_seqs_merged.csv')\nvalidation_seqs = pd.read_csv(f'{BASE_PATH}/validation_sequences.csv')\ntrain_labels = pd.read_csv('/kaggle/input/stanford-rna-3d-folding-pdb-parsed-data/all_labels_merged.csv', low_memory=False)\nvalidation_labels = pd.read_csv(f'{BASE_PATH}/validation_labels.csv')\nsample_submission = pd.read_csv(f'{BASE_PATH}/sample_submission.csv')\n\nif not IS_SCORING_RUN:\n    val_targets = set(validation_seqs['target_id'].values.tolist())\n    train_seqs = train_seqs[~train_seqs['target_id'].isin(val_targets)].reset_index(drop=True)\n    train_labels = train_labels[~train_labels['target_id'].isin(val_targets)].reset_index(drop=True)\n\nprint(f\"✓ Loaded {len(train_seqs)} training sequences\")\nprint(f\"✓ Loaded {len(validation_seqs)} validation sequences\") \nprint(f\"✓ Loaded {len(test_seqs)} test sequences\")\n\n# Cell 3: Process Training Labels to Coordinate Dictionary\ndef process_labels(labels_df, use_first_model_only=True):\n    \"\"\"\n    Process labels dataframe to create a dictionary mapping target_id to coordinates.\n    Vectorized implementation for improved performance.\n    \n    Args:\n        labels_df: DataFrame with ID, resid, x_1, y_1, z_1, etc.\n        use_first_model_only: If True, only extract first model coordinates\n        \n    Returns:\n        Dictionary mapping target_id to numpy array of coordinates\n    \"\"\"\n    print(\"Extracting target IDs...\")\n    # Vectorized target_id extraction\n    labels_df = labels_df.copy()\n    labels_df['target_id'] = labels_df['ID'].str.rsplit('_', n=1).str[0]\n    \n    # Sort once for all groups\n    labels_df = labels_df.sort_values(['target_id', 'resid'])\n    \n    # Vectorized coordinate extraction\n    print(\"Extracting coordinates...\")\n    coord_cols = ['x_1', 'y_1', 'z_1']\n    \n    # Replace placeholder values with NaN in one operation\n    coords_array = labels_df[coord_cols].values.copy()\n    coords_array[coords_array < -1e6] = np.nan\n    labels_df[coord_cols] = coords_array\n    \n    # Group and convert to dictionary with progress bar\n    print(\"Grouping by target_id...\")\n    coords_dict = {}\n    \n    grouped = labels_df.groupby('target_id', sort=False)\n    for target_id, group in tqdm(grouped, desc=\"Processing structures\", total=len(grouped)):\n        # Directly extract coordinates as numpy array (already sorted)\n        coords_dict[target_id] = group[coord_cols].values\n    \n    return coords_dict\n\n\ntrain_coords_dict = process_labels(train_labels)\nvalid_coords_dict = process_labels(validation_labels)\n\n# Cell 4: Sequence Alignment and Template Finding\ndef get_alignment_score(query_seq, template_seq):\n    alignments = ALIGNER.align(query_seq, template_seq)\n    best_alignment = next(iter(alignments), None)\n    \n    if best_alignment is None:\n        return 0.0\n    \n    # Normalize score by theoretical max (perfect match of shorter sequence)\n    max_possible = 2.1 * min(len(query_seq), len(template_seq))\n    normalized_score = best_alignment.score / max_possible\n    \n    return min(normalized_score, 1.0)  # Cap at 1.0\n\n\ndef get_aligned_sequences(query_seq, template_seq):\n    alignments = ALIGNER.align(query_seq, template_seq)\n    best_alignment = next(iter(alignments), None)\n    \n    if best_alignment is None:\n        return None, None\n    \n    # Use the robust method that builds from alignment coordinates\n    return build_aligned_sequences(query_seq, template_seq, best_alignment)\n\n\ndef build_aligned_sequences(query_seq, template_seq, alignment):\n    \"\"\"\n    Build aligned sequences with gaps from alignment coordinates.\n    Uses alignment.aligned property which returns tuples of (start, end) ranges.\n    \"\"\"\n    # Get aligned blocks: alignment.aligned returns ((query_ranges), (template_ranges))\n    query_ranges, template_ranges = alignment.aligned\n    \n    # If no aligned blocks, return None\n    if len(query_ranges) == 0:\n        return None, None\n    \n    aligned_query = []\n    aligned_template = []\n    \n    query_pos = 0\n    template_pos = 0\n    \n    for (q_start, q_end), (t_start, t_end) in zip(query_ranges, template_ranges):\n        # Add gaps for unaligned query residues (query has residues, template doesn't)\n        while query_pos < q_start:\n            aligned_query.append(query_seq[query_pos])\n            aligned_template.append('-')\n            query_pos += 1\n        \n        # Add gaps for unaligned template residues (template has residues, query doesn't)\n        while template_pos < t_start:\n            aligned_query.append('-')\n            aligned_template.append(template_seq[template_pos])\n            template_pos += 1\n        \n        # Add aligned region (both have residues)\n        block_len = q_end - q_start  # Should equal t_end - t_start\n        for i in range(block_len):\n            aligned_query.append(query_seq[q_start + i])\n            aligned_template.append(template_seq[t_start + i])\n        \n        query_pos = q_end\n        template_pos = t_end\n    \n    # Add any remaining unaligned query residues at the end\n    while query_pos < len(query_seq):\n        aligned_query.append(query_seq[query_pos])\n        aligned_template.append('-')\n        query_pos += 1\n    \n    # Add any remaining unaligned template residues at the end\n    while template_pos < len(template_seq):\n        aligned_query.append('-')\n        aligned_template.append(template_seq[template_pos])\n        template_pos += 1\n    \n    return ''.join(aligned_query), ''.join(aligned_template)\n\n\ndef find_similar_sequences(query_seq, train_seqs_df, train_coords_dict, \n                          temporal_cutoff=None, top_n=5):\n    \"\"\"\n    Find sequences in the training data similar to the query sequence.\n    Uses modern PairwiseAligner for sequence comparison.\n    \n    Args:\n        query_seq: The RNA sequence to find templates for\n        train_seqs_df: DataFrame containing training sequences\n        train_coords_dict: Dictionary mapping target_ids to their 3D coordinates\n        temporal_cutoff: Only consider training sequences published before this date\n        top_n: Number of top templates to return\n        \n    Returns:\n        List of (target_id, sequence, similarity_score, coordinates) tuples\n    \"\"\"\n    similar_seqs = []\n    \n    # Filter training sequences by temporal cutoff if provided\n    if temporal_cutoff:\n        filtered_train_seqs = train_seqs_df[train_seqs_df['temporal_cutoff'] < temporal_cutoff]\n    else:\n        filtered_train_seqs = train_seqs_df\n    \n    for _, row in filtered_train_seqs.iterrows():\n        target_id = row['target_id']\n        train_seq = row['sequence']\n        \n        # Skip if coordinates not available\n        if target_id not in train_coords_dict:\n            continue\n        \n        # Skip if sequence length difference is too large (>50%)\n        len_diff = abs(len(train_seq) - len(query_seq)) / max(len(train_seq), len(query_seq))\n        if len_diff > 0.5:\n            continue\n        \n        # Calculate similarity score using new aligner\n        similarity_score = get_alignment_score(query_seq, train_seq)\n        \n        similar_seqs.append((target_id, train_seq, similarity_score, train_coords_dict[target_id]))\n    \n    # Sort by similarity score (higher is better)\n    similar_seqs.sort(key=lambda x: x[2], reverse=True)\n    return similar_seqs[:top_n]\n\n# Cell 5: Template Adaptation Functions\ndef adapt_template_to_query(query_seq, template_seq, template_coords):\n    # Get aligned sequences\n    aligned_query, aligned_template = get_aligned_sequences(query_seq, template_seq)\n    \n    if aligned_query is None or aligned_template is None:\n        return adapt_template_simple(query_seq, template_seq, template_coords)\n    \n    # Initialize coordinates for query sequence\n    query_coords = np.zeros((len(query_seq), 3))\n    query_coords.fill(np.nan)\n    \n    # Map template coordinates to query based on alignment\n    query_idx = 0\n    template_idx = 0\n    \n    for i in range(len(aligned_query)):\n        query_char = aligned_query[i]\n        template_char = aligned_template[i]\n        \n        if query_char != '-' and template_char != '-':\n            # Both aligned - copy template coordinate to query\n            if query_idx < len(query_seq) and template_idx < len(template_coords):\n                # Handle NaN coordinates in template\n                if not np.any(np.isnan(template_coords[template_idx])):\n                    query_coords[query_idx] = template_coords[template_idx]\n            template_idx += 1\n            query_idx += 1\n        elif query_char != '-' and template_char == '-':\n            # Gap in template - query residue has no template coord\n            query_idx += 1\n        elif query_char == '-' and template_char != '-':\n            # Gap in query - skip template residue\n            template_idx += 1\n    \n    # Fill in gaps by interpolation\n    query_coords = fill_coordinate_gaps(query_coords)\n    \n    return query_coords\n\n\ndef adapt_template_simple(query_seq, template_seq, template_coords):\n    \"\"\"Simple template adaptation without Biopython alignment.\"\"\"\n    query_coords = np.zeros((len(query_seq), 3))\n    \n    # Simple mapping based on position\n    scale = len(template_coords) / len(query_seq)\n    for i in range(len(query_seq)):\n        template_idx = int(i * scale)\n        template_idx = min(template_idx, len(template_coords) - 1)\n        if not np.any(np.isnan(template_coords[template_idx])):\n            query_coords[i] = template_coords[template_idx]\n        else:\n            query_coords[i] = [np.nan, np.nan, np.nan]\n    \n    # Fill gaps\n    query_coords = fill_coordinate_gaps(query_coords)\n    return query_coords\n\n\ndef fill_coordinate_gaps(coords):\n    \"\"\"Fill NaN gaps in coordinates by interpolation.\"\"\"\n    n = len(coords)\n    typical_step = 4.0\n    \n    # First pass: interpolate between valid points\n    for i in range(n):\n        if np.isnan(coords[i, 0]):\n            prev_valid = next((j for j in range(i-1, -1, -1) if not np.isnan(coords[j, 0])), -1)\n            next_valid = next((j for j in range(i+1, n) if not np.isnan(coords[j, 0])), -1)\n            \n            if prev_valid >= 0 and next_valid >= 0:\n                weight = (i - prev_valid) / (next_valid - prev_valid)\n                coords[i] = (1 - weight) * coords[prev_valid] + weight * coords[next_valid]\n    \n    # Second pass: handle remaining NaNs at edges\n    for i in range(n):\n        if np.isnan(coords[i, 0]):\n            if i == 0:\n                first_valid = next((j for j in range(1, n) if not np.isnan(coords[j, 0])), -1)\n                if first_valid >= 0:\n                    for j in range(first_valid - 1, -1, -1):\n                        direction = np.random.normal(0, 1, 3)\n                        direction = direction / (np.linalg.norm(direction) + 1e-10) * typical_step\n                        coords[j] = coords[j + 1] - direction\n                else:\n                    coords = generate_basic_structure_coords(n)\n                    break\n            else:\n                prev_valid = next((j for j in range(i-1, -1, -1) if not np.isnan(coords[j, 0])), -1)\n                if prev_valid >= 0:\n                    direction = np.random.normal(0, 1, 3)\n                    direction = direction / (np.linalg.norm(direction) + 1e-10) * typical_step\n                    coords[i] = coords[prev_valid] + direction\n    \n    # Final cleanup\n    coords = np.nan_to_num(coords)\n    return coords\n\n\ndef generate_basic_structure(sequence):\n    \"\"\"Generate a simple helical structure.\"\"\"\n    return generate_basic_structure_coords(len(sequence))\n\n\ndef generate_basic_structure_coords(n_residues):\n    \"\"\"Generate basic helical coordinates.\"\"\"\n    coords = np.zeros((n_residues, 3))\n    for i in range(n_residues):\n        angle = i * 0.6\n        coords[i] = [10.0 * np.cos(angle), 10.0 * np.sin(angle), i * 2.5]\n    return coords\n\n# Cell 6: RNA Geometric Constraints\ndef adaptive_rna_constraints(coordinates, sequence, confidence=1.0):\n    \"\"\"\n    Apply RNA geometric constraints with adaptive strength based on confidence.\n    \"\"\"\n    refined_coords = coordinates.copy()\n    n_residues = len(sequence)\n    \n    constraint_strength = 0.8 * (1.0 - min(confidence, 0.8))\n    \n    # Sequential distance constraints\n    seq_min_dist, seq_max_dist = 5.5, 6.5\n    \n    for i in range(n_residues - 1):\n        current_pos = refined_coords[i]\n        next_pos = refined_coords[i + 1]\n        current_dist = np.linalg.norm(next_pos - current_pos)\n        \n        if current_dist < seq_min_dist or current_dist > seq_max_dist:\n            target_dist = (seq_min_dist + seq_max_dist) / 2\n            direction = next_pos - current_pos\n            direction = direction / (np.linalg.norm(direction) + 1e-10)\n            adjustment = (target_dist - current_dist) * constraint_strength\n            refined_coords[i + 1] = current_pos + direction * (current_dist + adjustment)\n    \n    # Steric clash prevention\n    min_allowed_distance = 3.8\n    dist_matrix = distance_matrix(refined_coords, refined_coords)\n    severe_clashes = np.where((dist_matrix < min_allowed_distance) & (dist_matrix > 0))\n    \n    for idx in range(len(severe_clashes[0])):\n        i, j = severe_clashes[0][idx], severe_clashes[1][idx]\n        if abs(i - j) <= 1 or i >= j:\n            continue\n        \n        pos_i, pos_j = refined_coords[i], refined_coords[j]\n        current_dist = dist_matrix[i, j]\n        direction = pos_j - pos_i\n        direction = direction / (np.linalg.norm(direction) + 1e-10)\n        adjustment = (min_allowed_distance - current_dist) * constraint_strength\n        refined_coords[i] = pos_i - direction * (adjustment / 2)\n        refined_coords[j] = pos_j + direction * (adjustment / 2)\n    \n    return refined_coords\n\n# Cell 7: De Novo Structure Generation\ndef generate_rna_structure(sequence, seed=None):\n    \"\"\"Generate a more realistic RNA structure prediction.\"\"\"\n    if seed is not None:\n        np.random.seed(seed)\n        random.seed(seed)\n    \n    n_residues = len(sequence)\n    coordinates = np.zeros((n_residues, 3))\n    \n    # Initialize first residues\n    for i in range(min(3, n_residues)):\n        angle = i * 0.6\n        coordinates[i] = [10.0 * np.cos(angle), 10.0 * np.sin(angle), i * 2.5]\n    \n    current_direction = np.array([0.0, 0.0, 1.0])\n    complementary = {'G': 'C', 'C': 'G', 'A': 'U', 'U': 'A'}\n    \n    for i in range(3, n_residues):\n        current_base = sequence[i]\n        has_pair = False\n        pair_idx = -1\n        \n        window_size = min(i, 15)\n        for j in range(i - window_size, i):\n            if j >= 0 and sequence[j] == complementary.get(current_base, 'X'):\n                has_pair = True\n                pair_idx = j\n                break\n        \n        if has_pair and i - pair_idx <= 10 and random.random() < 0.7:\n            pair_pos = coordinates[pair_idx]\n            random_offset = np.random.normal(0, 1, 3) * 2.0\n            base_pair_distance = 10.0 + random.uniform(-1.0, 1.0)\n            \n            center = np.mean(coordinates[:i], axis=0)\n            direction = center - pair_pos\n            direction = direction / (np.linalg.norm(direction) + 1e-10)\n            \n            coordinates[i] = pair_pos + direction * base_pair_distance + random_offset\n            current_direction = np.random.normal(0, 0.3, 3)\n            current_direction = current_direction / (np.linalg.norm(current_direction) + 1e-10)\n        else:\n            if random.random() < 0.3:\n                angle = random.uniform(0.2, 0.6)\n                axis = np.random.normal(0, 1, 3)\n                axis = axis / (np.linalg.norm(axis) + 1e-10)\n                rotation = R.from_rotvec(angle * axis)\n                current_direction = rotation.apply(current_direction)\n            else:\n                current_direction += np.random.normal(0, 0.15, 3)\n                current_direction = current_direction / (np.linalg.norm(current_direction) + 1e-10)\n            \n            step_size = random.uniform(3.5, 4.5)\n            coordinates[i] = coordinates[i - 1] + step_size * current_direction\n    \n    return coordinates\n\n# Cell 8: Main Prediction Function\ndef predict_rna_structures(sequence, target_id, train_seqs_df, train_coords_dict, \n                          n_predictions=5, temporal_cutoff=None):\n    \"\"\"Generate multiple structure predictions for an RNA sequence.\"\"\"\n    predictions = []\n    \n    similar_seqs = find_similar_sequences(\n        sequence, train_seqs_df, train_coords_dict,\n        temporal_cutoff=temporal_cutoff, top_n=n_predictions\n    )\n    \n    # Use templates if found\n    if similar_seqs:\n        for template_id, template_seq, similarity, template_coords in similar_seqs:\n            adapted_coords = adapt_template_to_query(sequence, template_seq, template_coords)\n            \n            if adapted_coords is not None:\n                refined_coords = adaptive_rna_constraints(adapted_coords, sequence, confidence=similarity)\n                random_scale = max(0.05, 0.8 - similarity)\n                randomized_coords = refined_coords + np.random.normal(0, random_scale, refined_coords.shape)\n                predictions.append(randomized_coords)\n                \n                if len(predictions) >= n_predictions:\n                    break\n    \n    # Fill remaining with de novo structures\n    while len(predictions) < n_predictions:\n        # Create a stable integer seed from target_id + offset\n        # hash() is non-deterministic in Python 3+, use hashlib instead\n        unique_str = f\"{target_id}_{len(predictions)}\"\n        stable_hash = int(hashlib.sha256(unique_str.encode('utf-8')).hexdigest(), 16)\n        seed_value = stable_hash % (2**32 - 1)  # Keep within valid range\n        \n        de_novo_coords = generate_rna_structure(sequence, seed=seed_value)\n        refined_de_novo = adaptive_rna_constraints(de_novo_coords, sequence, confidence=0.2)\n        predictions.append(refined_de_novo)\n    \n    return predictions[:n_predictions]\n\n# Cell 9: Generate Predictions (Optimized with Parallel)\ndef process_target_row(row, train_seqs_df, train_coords_dict):\n    \"\"\"\n    Helper function to process a single sequence row in parallel.\n    Encapsulates the prediction and formatting logic for one target.\n    \"\"\"\n    target_id = row['target_id']\n    sequence = row['sequence']\n    \n    # temporal_cutoff = row.get('temporal_cutoff', None)\n    temporal_cutoff = None\n    \n    # --- CRITICAL FIX FOR REPRODUCIBILITY ---\n    # Create a deterministic seed from the target_id\n    stable_hash = int(hashlib.sha256(target_id.encode('utf-8')).hexdigest(), 16)\n    row_seed = stable_hash % (2**32 - 1)\n    \n    # Seed the random number generators for this specific worker/task\n    np.random.seed(row_seed)\n    random.seed(row_seed)\n    # ----------------------------------------\n    \n    # Run prediction\n    predictions = predict_rna_structures(\n        sequence, target_id, train_seqs_df, train_coords_dict,\n        n_predictions=5, temporal_cutoff=temporal_cutoff\n    )\n    \n    # Format results\n    row_predictions = []\n    for j in range(len(sequence)):\n        pred_item = {\n            'ID': f\"{target_id}_{j+1}\",\n            'resname': sequence[j],\n            'resid': j + 1\n        }\n        for i in range(5):\n            pred_item[f'x_{i+1}'] = predictions[i][j][0]\n            pred_item[f'y_{i+1}'] = predictions[i][j][1]\n            pred_item[f'z_{i+1}'] = predictions[i][j][2]\n        row_predictions.append(pred_item)\n        \n    return row_predictions\n\ndef generate_predictions_for_dataset(seqs_df, train_seqs_df, train_coords_dict, dataset_name=\"dataset\"):\n    \"\"\"Generate predictions for a given sequence dataset using parallel processing.\"\"\"\n    start_time = time.time()\n    total_targets = len(seqs_df)\n    \n    print(f\"\\n=== Generating Predictions for {dataset_name} ({total_targets} sequences) ===\")\n    \n    # Use joblib to parallelize the loop\n    # We iterate over rows and process each independently\n    # Backend 'loky' is safer for isolation\n    results = Parallel(n_jobs=-1, backend='loky')(\n        delayed(process_target_row)(row, train_seqs_df, train_coords_dict)\n        for _, row in tqdm(seqs_df.iterrows(), total=total_targets, desc=\"Computing structures\")\n    )\n    \n    # Flatten the list of lists into a single list of dictionaries\n    all_predictions = [item for sublist in results for item in sublist]\n    \n    # Create DataFrame\n    df = pd.DataFrame(all_predictions)\n    \n    # Ensure correct column order\n    column_order = ['ID', 'resname', 'resid']\n    for i in range(1, 6):\n        for coord in ['x', 'y', 'z']:\n            column_order.append(f'{coord}_{i}')\n            \n    # Select only columns that exist\n    df = df[[c for c in column_order if c in df.columns]]\n    \n    print(f\"\\nGenerated predictions for {total_targets} sequences\")\n    print(f\"Total runtime: {time.time() - start_time:.1f} seconds\")\n    print(f\"Output shape: {df.shape}\")\n    \n    return df\n\n\n# Cell 10: Generate Predictions for TEST Set (Submission)\n# Generate test predictions for submission\ntest_pred_df = generate_predictions_for_dataset(\n    seqs_df=test_seqs,\n    # Use combined data from train + validation at scoring run\n    train_seqs_df=pd.concat([train_seqs, validation_seqs]) if IS_SCORING_RUN else train_seqs,\n    train_coords_dict={**train_coords_dict, **valid_coords_dict} if IS_SCORING_RUN else train_coords_dict,\n    dataset_name=\"Test\"\n)\ntest_pred_df.head()\n\n# Cell 11: Save Submission\ntest_pred_df.to_csv('submission_tbm.csv', index=False)\nprint(\"\\n=== Submission file saved ===\")\nprint(f\"submission.csv shape: {test_pred_df.shape}\")\nprint(f\"\\nFirst few rows:\")\ntest_pred_df.head(10)","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:16:34.101901Z","iopub.status.busy":"2026-02-16T19:16:34.101583Z","iopub.status.idle":"2026-02-16T19:16:34.114955Z","shell.execute_reply":"2026-02-16T19:16:34.114187Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.024833,"end_time":"2026-02-16T19:16:34.116317","exception":false,"start_time":"2026-02-16T19:16:34.091484","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"7d8da799","cell_type":"code","source":"!python gen_submission_tbm.py","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:16:34.134954Z","iopub.status.busy":"2026-02-16T19:16:34.134734Z","iopub.status.idle":"2026-02-16T19:32:30.153305Z","shell.execute_reply":"2026-02-16T19:32:30.152403Z"},"papermill":{"duration":956.029861,"end_time":"2026-02-16T19:32:30.155086","exception":false,"start_time":"2026-02-16T19:16:34.125225","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"bf3e15b4","cell_type":"markdown","source":"$$ $$","metadata":{"papermill":{"duration":0.010908,"end_time":"2026-02-16T19:32:30.177243","exception":false,"start_time":"2026-02-16T19:32:30.166335","status":"completed"},"tags":[]}},{"id":"ebae4eda","cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom tqdm.auto import tqdm\nfrom scipy.spatial.transform import Rotation as R\nimport random\nfrom Bio import pairwise2\nfrom Bio.Seq import Seq\nimport time\nfrom scipy.spatial import distance_matrix\nimport warnings\n\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:32:30.200213Z","iopub.status.busy":"2026-02-16T19:32:30.19994Z","iopub.status.idle":"2026-02-16T19:32:30.907594Z","shell.execute_reply":"2026-02-16T19:32:30.906856Z"},"papermill":{"duration":0.721303,"end_time":"2026-02-16T19:32:30.909246","exception":false,"start_time":"2026-02-16T19:32:30.187943","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"cefd9cd5","cell_type":"code","source":"DATA_PATH = '/kaggle/input/stanford-rna-3d-folding-2/'\n\nprint(f\"[+] Read Files\")\ntrain_seqs = pd.read_csv(DATA_PATH + 'train_sequences.csv')\ntest_seqs = pd.read_csv(DATA_PATH + 'test_sequences.csv')\ntrain_labels = pd.read_csv(DATA_PATH + 'train_labels.csv')\n\ntry:\n    validation_seqs = pd.read_csv(DATA_PATH + 'validation_sequences.csv')\n    validation_labels = pd.read_csv(DATA_PATH + 'validation_labels.csv')\n    print(\"Validation data found and will be combined with train data.\")\n    \n    combined_seqs = pd.concat([train_seqs, validation_seqs], ignore_index=True)\n    combined_labels = pd.concat([train_labels, validation_labels], ignore_index=True)\n    \nexcept FileNotFoundError:\n    print(\"Validation data not found, using only train data.\")\n    combined_seqs = train_seqs\n    combined_labels = train_labels","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:32:30.933078Z","iopub.status.busy":"2026-02-16T19:32:30.93255Z","iopub.status.idle":"2026-02-16T19:32:45.43031Z","shell.execute_reply":"2026-02-16T19:32:45.429713Z"},"papermill":{"duration":14.511242,"end_time":"2026-02-16T19:32:45.431915","exception":false,"start_time":"2026-02-16T19:32:30.920673","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"4d853541","cell_type":"code","source":"def process_labels(labels_df):\n    coords_dict = {}\n    for id_prefix, group in tqdm(labels_df.groupby(lambda x: labels_df['ID'][x].rsplit('_', 1)[0])):\n        coords = [group.sort_values('resid')[['x_1', 'y_1', 'z_1']].values]\n        coords_dict[id_prefix] = coords[0]\n    return coords_dict\n\ncombined_coords_dict = process_labels(combined_labels)","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:32:45.455901Z","iopub.status.busy":"2026-02-16T19:32:45.455364Z","iopub.status.idle":"2026-02-16T19:33:25.661265Z","shell.execute_reply":"2026-02-16T19:33:25.660611Z"},"papermill":{"duration":40.219436,"end_time":"2026-02-16T19:33:25.662734","exception":false,"start_time":"2026-02-16T19:32:45.443298","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"97de3d4c","cell_type":"markdown","source":"### All Functions","metadata":{"papermill":{"duration":0.011118,"end_time":"2026-02-16T19:33:25.68562","exception":false,"start_time":"2026-02-16T19:33:25.674502","status":"completed"},"tags":[]}},{"id":"fa0d9682","cell_type":"code","source":"def find_similar_sequences(query_seq, train_seqs_df, train_coords_dict, top_n=5):\n    similar_seqs = []\n    query_seq_obj = Seq(query_seq)\n    \n    for _, row in train_seqs_df.iterrows():\n        target_id, train_seq = row['target_id'], row['sequence']\n        if target_id not in train_coords_dict: continue\n        \n        if abs(len(train_seq) - len(query_seq)) / max(len(train_seq), len(query_seq)) > 0.3: continue\n        \n        alignments = pairwise2.align.globalms(query_seq_obj, train_seq, 2, -1, -3, -0.1, one_alignment_only=True)\n        \n        if alignments:\n            score = alignments[0].score / (2 * min(len(query_seq), len(train_seq)))\n            similar_seqs.append((target_id, train_seq, score, train_coords_dict[target_id]))\n    \n    similar_seqs.sort(key=lambda x: x[2], reverse=True)\n    return similar_seqs[:top_n]\n\ndef adaptive_rna_constraints(coordinates, sequence, confidence=1.0):\n    refined_coords = coordinates.copy()\n    n_residues = len(sequence)\n    \n    constraint_strength = 0.4 * (1.0 - min(confidence, 0.9))\n    \n    seq_min_dist, seq_max_dist = 5.5, 6.5\n    \n    for i in range(n_residues - 1):\n        dist = np.linalg.norm(refined_coords[i+1] - refined_coords[i])\n        if dist < seq_min_dist or dist > seq_max_dist:\n            target_dist = 6.0\n            direction = (refined_coords[i+1] - refined_coords[i]) / (dist + 1e-10)\n            adjustment = (target_dist - dist) * constraint_strength\n            refined_coords[i+1] = refined_coords[i+1] + direction * adjustment\n            \n    return refined_coords\n\ndef adapt_template_to_query(query_seq, template_seq, template_coords):\n    alignments = pairwise2.align.globalms(Seq(query_seq), Seq(template_seq), 2, -1, -3, -0.1, one_alignment_only=True)\n    if not alignments: return np.zeros((len(query_seq), 3))\n    \n    a_q, a_t = alignments[0].seqA, alignments[0].seqB\n    new_coords = np.full((len(query_seq), 3), np.nan)\n    q_idx, t_idx = 0, 0\n    for char_q, char_t in zip(a_q, a_t):\n        if char_q != '-' and char_t != '-':\n            if t_idx < len(template_coords): new_coords[q_idx] = template_coords[t_idx]\n            q_idx += 1; t_idx += 1\n        elif char_q != '-': q_idx += 1\n        elif char_t != '-': t_idx += 1\n\n    for i in range(len(new_coords)):\n        if np.isnan(new_coords[i, 0]):\n            prev_v = next((j for j in range(i-1, -1, -1) if not np.isnan(new_coords[j, 0])), -1)\n            next_v = next((j for j in range(i+1, len(new_coords)) if not np.isnan(new_coords[j, 0])), -1)\n            if prev_v >= 0 and next_v >= 0:\n                w = (i - prev_v) / (next_v - prev_v)\n                new_coords[i] = (1-w)*new_coords[prev_v] + w*new_coords[next_v]\n            elif prev_v >= 0: new_coords[i] = new_coords[prev_v] + [3, 0, 0]\n            elif next_v >= 0: new_coords[i] = new_coords[next_v] + [3, 0, 0]\n            else: new_coords[i] = [i*3, 0, 0]\n    return np.nan_to_num(new_coords)\n\ndef generate_rna_structure(sequence, seed=None):\n    if seed: np.random.seed(seed)\n    n = len(sequence)\n    coords = np.zeros((n, 3))\n    for i in range(1, n):\n        step_size = random.uniform(3.0, 5.0)\n        coords[i] = coords[i-1] + [step_size, 0, 0]\n    return coords\n\ndef predict_rna_structures(sequence, target_id, train_seqs_df, train_coords_dict, n_predictions=5):\n    predictions = []\n    similar_seqs = find_similar_sequences(sequence, train_seqs_df, train_coords_dict, top_n=n_predictions)\n    \n    if similar_seqs:\n        for i, (template_id, template_seq, similarity, template_coords) in enumerate(similar_seqs):\n            adapted = adapt_template_to_query(sequence, template_seq, template_coords)\n            refined = adaptive_rna_constraints(adapted, sequence, confidence=similarity)\n            \n            random_scale = max(0.05, (0.5 - similarity) * 0.15)\n            refined += np.random.normal(0, random_scale, refined.shape)\n            \n            predictions.append(refined)\n                \n    while len(predictions) < n_predictions:\n        predictions.append(generate_rna_structure(sequence, seed=len(predictions)))\n    \n    return predictions[:n_predictions]","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:33:25.710039Z","iopub.status.busy":"2026-02-16T19:33:25.709446Z","iopub.status.idle":"2026-02-16T19:33:25.724806Z","shell.execute_reply":"2026-02-16T19:33:25.724126Z"},"papermill":{"duration":0.029145,"end_time":"2026-02-16T19:33:25.726163","exception":false,"start_time":"2026-02-16T19:33:25.697018","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"96f87bc9","cell_type":"markdown","source":"### Generate Predictions","metadata":{"papermill":{"duration":0.010988,"end_time":"2026-02-16T19:33:25.748204","exception":false,"start_time":"2026-02-16T19:33:25.737216","status":"completed"},"tags":[]}},{"id":"3342a3b3","cell_type":"code","source":"all_predictions = []\nfor idx, row in tqdm(test_seqs.iterrows(), total = len(test_seqs)):\n    target_id, sequence = row['target_id'], row['sequence']\n\n    ## Public LeaderBoard\n    if IS_SCORING_RUN is None:\n        for j in range(len(sequence)):\n            res = {'ID': f\"{target_id}_{j+1}\", 'resname': sequence[j], 'resid': j+1}\n            for i in range(5):\n                res[f'x_{i+1}'], res[f'y_{i+1}'], res[f'z_{i+1}'] = (0.0,0.0,0.0)\n            all_predictions.append(res)\n        continue\n    #############\n    \n    preds = predict_rna_structures(sequence, target_id, combined_seqs, combined_coords_dict)\n    \n    for j in range(len(sequence)):\n        res = {'ID': f\"{target_id}_{j+1}\", 'resname': sequence[j], 'resid': j+1}\n        for i in range(5):\n            res[f'x_{i+1}'], res[f'y_{i+1}'], res[f'z_{i+1}'] = preds[i][j]\n                \n        all_predictions.append(res)","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:33:25.772075Z","iopub.status.busy":"2026-02-16T19:33:25.771353Z","iopub.status.idle":"2026-02-16T19:33:25.831937Z","shell.execute_reply":"2026-02-16T19:33:25.831225Z"},"papermill":{"duration":0.074116,"end_time":"2026-02-16T19:33:25.833516","exception":false,"start_time":"2026-02-16T19:33:25.7594","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"7c1a3fb2","cell_type":"code","source":"submission_df = pd.DataFrame(all_predictions)\nsubmission_df.to_csv('submission_long.csv', index=False)\nsubmission_df","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:33:25.857008Z","iopub.status.busy":"2026-02-16T19:33:25.856622Z","iopub.status.idle":"2026-02-16T19:33:25.980668Z","shell.execute_reply":"2026-02-16T19:33:25.979885Z"},"papermill":{"duration":0.137553,"end_time":"2026-02-16T19:33:25.982161","exception":false,"start_time":"2026-02-16T19:33:25.844608","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"94e27b52","cell_type":"markdown","source":"$$ $$","metadata":{"papermill":{"duration":0.011596,"end_time":"2026-02-16T19:33:26.00565","exception":false,"start_time":"2026-02-16T19:33:25.994054","status":"completed"},"tags":[]}},{"id":"28fb6bcc","cell_type":"code","source":"import os, sys\nimport pandas as pd\nimport numpy as np\nfrom scipy.spatial.transform import Rotation as R\nfrom scipy.spatial import distance_matrix\nimport random\nimport hashlib\nimport time\nimport warnings\nfrom joblib import Parallel, delayed\n\nfrom tqdm.auto import tqdm\nwarnings.filterwarnings('ignore')\nfrom Bio.Align import PairwiseAligner\nfrom Bio.Seq import Seq\n\nsub_tbm = pd.read_csv(\"/kaggle/working/submission_tbm.csv\")\nsub_tbm","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:33:26.030077Z","iopub.status.busy":"2026-02-16T19:33:26.02946Z","iopub.status.idle":"2026-02-16T19:33:26.095092Z","shell.execute_reply":"2026-02-16T19:33:26.094421Z"},"papermill":{"duration":0.079508,"end_time":"2026-02-16T19:33:26.096553","exception":false,"start_time":"2026-02-16T19:33:26.017045","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"6ae84975","cell_type":"code","source":"pwd","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:33:26.122067Z","iopub.status.busy":"2026-02-16T19:33:26.121803Z","iopub.status.idle":"2026-02-16T19:33:26.126053Z","shell.execute_reply":"2026-02-16T19:33:26.125437Z"},"papermill":{"duration":0.018037,"end_time":"2026-02-16T19:33:26.127381","exception":false,"start_time":"2026-02-16T19:33:26.109344","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"1f9e3e56","cell_type":"markdown","source":"continue RNAPro\n---  \n\n","metadata":{"papermill":{"duration":0.011854,"end_time":"2026-02-16T19:33:26.150776","exception":false,"start_time":"2026-02-16T19:33:26.138922","status":"completed"},"tags":[]}},{"id":"6c77f1be","cell_type":"code","source":"cd RNAPro","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:33:26.175068Z","iopub.status.busy":"2026-02-16T19:33:26.174847Z","iopub.status.idle":"2026-02-16T19:33:26.179051Z","shell.execute_reply":"2026-02-16T19:33:26.178473Z"},"papermill":{"duration":0.01781,"end_time":"2026-02-16T19:33:26.180296","exception":false,"start_time":"2026-02-16T19:33:26.162486","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"17907bc8","cell_type":"code","source":"!python preprocess/convert_templates_to_pt_files.py --input_csv /kaggle/working/submission_tbm.csv --output_name templates.pt","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:33:26.205052Z","iopub.status.busy":"2026-02-16T19:33:26.204837Z","iopub.status.idle":"2026-02-16T19:33:31.934759Z","shell.execute_reply":"2026-02-16T19:33:31.933915Z"},"papermill":{"duration":5.744396,"end_time":"2026-02-16T19:33:31.936452","exception":false,"start_time":"2026-02-16T19:33:26.192056","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"fe92d57b","cell_type":"code","source":"DIST = \"/kaggle/working/RNAPro/release_data/ccd_cache/\"\n!mkdir -p $DIST","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:33:31.962504Z","iopub.status.busy":"2026-02-16T19:33:31.962038Z","iopub.status.idle":"2026-02-16T19:33:32.127176Z","shell.execute_reply":"2026-02-16T19:33:32.126192Z"},"papermill":{"duration":0.179988,"end_time":"2026-02-16T19:33:32.128913","exception":false,"start_time":"2026-02-16T19:33:31.948925","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"63c130ba","cell_type":"code","source":"# !python preprocess/gen_ccd_cache.py -c $DIST","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:33:32.155245Z","iopub.status.busy":"2026-02-16T19:33:32.154959Z","iopub.status.idle":"2026-02-16T19:33:32.158388Z","shell.execute_reply":"2026-02-16T19:33:32.157856Z"},"papermill":{"duration":0.018389,"end_time":"2026-02-16T19:33:32.159681","exception":false,"start_time":"2026-02-16T19:33:32.141292","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"6a815de4","cell_type":"code","source":"!cp /kaggle/input/protenix-checkpoints-jan-20-2026/components.cif $DIST\n!cp /kaggle/input/protenix-checkpoints-jan-20-2026/components.cif.rdkit_mol.pkl $DIST","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:33:32.184972Z","iopub.status.busy":"2026-02-16T19:33:32.184744Z","iopub.status.idle":"2026-02-16T19:33:39.938254Z","shell.execute_reply":"2026-02-16T19:33:39.937355Z"},"papermill":{"duration":7.768313,"end_time":"2026-02-16T19:33:39.93995","exception":false,"start_time":"2026-02-16T19:33:32.171637","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"ee1e265a","cell_type":"markdown","source":"Inference  \n---","metadata":{"papermill":{"duration":0.012028,"end_time":"2026-02-16T19:33:39.964573","exception":false,"start_time":"2026-02-16T19:33:39.952545","status":"completed"},"tags":[]}},{"id":"13a73a9a","cell_type":"code","source":"import pandas as pd\ndf = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding-2/test_sequences.csv\")\nif not IS_SCORING_RUN:\n    df = df.head(5)\n    \ndf.to_csv('/kaggle/working/sample_sequences.csv', index=False)","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:33:39.990103Z","iopub.status.busy":"2026-02-16T19:33:39.989493Z","iopub.status.idle":"2026-02-16T19:33:39.998085Z","shell.execute_reply":"2026-02-16T19:33:39.997554Z"},"papermill":{"duration":0.023099,"end_time":"2026-02-16T19:33:39.99948","exception":false,"start_time":"2026-02-16T19:33:39.976381","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"3a340860","cell_type":"code","source":"%%writefile runner/inference.py\n# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.\n# SPDX-License-Identifier: Apache-2.0\n#\n# Licensed under the Apache License, Version 2.0 (the \"License\");\n# you may not use this file except in compliance with the License.\n# You may obtain a copy of the License at\n#\n#     http://www.apache.org/licenses/LICENSE-2.0\n#\n# Unless required by applicable law or agreed to in writing, software\n# distributed under the License is distributed on an \"AS IS\" BASIS,\n# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n# See the License for the specific language governing permissions and\n# limitations under the License.\n\nimport os\nimport shutil\nimport logging\nimport traceback\nimport warnings\nimport argparse\nfrom contextlib import nullcontext\nfrom os.path import join as opjoin\nfrom typing import Any, Mapping\n\nimport json\nimport torch\nimport pandas as pd\nimport numpy as np\nfrom biotite.structure.io import pdbx\n\nfrom configs.configs_base import configs as configs_base\nfrom configs.configs_data import data_configs\nfrom configs.configs_inference import inference_configs\nfrom runner.dumper import DataDumper\n\nfrom rnapro.config import parse_sys_args\nfrom rnapro.config.config import ConfigManager, ArgumentNotSet\nfrom rnapro.data.infer_data_pipeline import get_inference_dataloader\nfrom rnapro.model.RNAPro import RNAPro\nfrom rnapro.utils.distributed import DIST_WRAPPER\nfrom rnapro.utils.seed import seed_everything\nfrom rnapro.utils.torch_utils import to_device\n\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\n\nlogger = logging.getLogger(__name__)\n\n# Silence all info logging\nlogging.basicConfig(level=logging.WARNING)\n# Silence dataloader logging specifically\nlogging.getLogger(\"rnapro.data\").setLevel(logging.WARNING)\nlogging.getLogger(\"rnapro\").setLevel(logging.WARNING)\n\n\ndef parse_configs(\n    configs: dict, arg_str: str = None, fill_required_with_null: bool = False\n):\n    \"\"\"\n    Parses and merges configuration settings from a dictionary and command-line arguments.\n\n    Args:\n        configs (dict): A dictionary containing initial configuration settings.\n        arg_str (str, optional): A string representing command-line arguments. Defaults to None.\n        fill_required_with_null (bool, optional):\n            A boolean flag indicating whether required values should be filled with `None` if not provided. Defaults to False.\n\n    Returns:\n        ConfigDict: The merged configuration dictionary.\n    \"\"\"\n    manager = ConfigManager(configs, fill_required_with_null=fill_required_with_null)\n    parser = argparse.ArgumentParser()\n\n    # This is new\n    parser.add_argument(\n        \"--max_len\",\n        type=int,\n        default=10000,\n        required=False,\n        help=\"Maximum length of the sequence. Longer sequences will be skipped during inference\"\n    )\n    parser.add_argument(\n        \"--n_templates_inf\",\n        type=int,\n        default=5,\n        required=False,\n        help=\"Number of templates to use during inference\"\n    )\n\n    # Register arguments\n    for key, (\n        dtype,\n        default_value,\n        allow_none,\n        required,\n    ) in manager.config_infos.items():\n        # All config use str type, strings will be converted to real dtype later\n        parser.add_argument(\n            \"--\" + key, type=str, default=ArgumentNotSet(), required=required\n        )\n    # Merge user commandline pargs with default ones\n    merged_configs = manager.merge_configs(\n        vars(parser.parse_args(arg_str.split())) if arg_str else {}\n    )\n\n    max_len = parser.parse_args(arg_str.split()).max_len\n    merged_configs.max_len = max_len\n\n    n_templates_inf = parser.parse_args(arg_str.split()).n_templates_inf\n    merged_configs.n_templates_inf = n_templates_inf\n\n    return merged_configs\n\n\nclass dotdict(dict):\n    __setattr__ = dict.__setitem__\n    __delattr__ = dict.__delitem__\n\n    def __getattr__(self, name):\n        try:\n            return self[name]\n        except KeyError:\n            raise AttributeError(name)\n\n\nclass InferenceRunner(object):\n    def __init__(self, configs: Any) -> None:\n        self.configs = configs\n        self.init_env()\n        self.init_basics()\n        self.init_model()\n        self.load_checkpoint()\n        self.init_dumper(\n            need_atom_confidence=configs.need_atom_confidence,\n            sorted_by_ranking_score=configs.sorted_by_ranking_score,\n        )\n\n    def init_env(self) -> None:\n        self.print(\n            f\"Distributed environment: world size: {DIST_WRAPPER.world_size}, \"\n            + f\"global rank: {DIST_WRAPPER.rank}, local rank: {DIST_WRAPPER.local_rank}\"\n        )\n        self.use_cuda = torch.cuda.device_count() > 0\n        if self.use_cuda:\n            self.device = torch.device(\"cuda:{}\".format(DIST_WRAPPER.local_rank))\n            os.environ[\"CUDA_DEVICE_ORDER\"] = \"PCI_BUS_ID\"\n            all_gpu_ids = \",\".join(str(x) for x in range(torch.cuda.device_count()))\n            devices = os.getenv(\"CUDA_VISIBLE_DEVICES\", all_gpu_ids)\n            logging.info(\n                f\"LOCAL_RANK: {DIST_WRAPPER.local_rank} - CUDA_VISIBLE_DEVICES: [{devices}]\"\n            )\n            torch.cuda.set_device(self.device)\n        else:\n            self.device = torch.device(\"cpu\")\n        if self.configs.use_deepspeed_evo_attention:\n            env = os.getenv(\"CUTLASS_PATH\", None)\n            self.print(f\"env: {env}\")\n            assert (\n                env is not None\n            ), \"if use ds4sci, set `CUTLASS_PATH` env as https://www.deepspeed.ai/tutorials/ds4sci_evoformerattention/\"\n            if env is not None:\n                logging.info(\n                    \"The kernels will be compiled when DS4Sci_EvoformerAttention is called for the first time.\"\n                )\n        use_fastlayernorm = os.getenv(\"LAYERNORM_TYPE\", None)\n        if use_fastlayernorm == \"fast_layernorm\":\n            logging.info(\n                \"The kernels will be compiled when fast_layernorm is called for the first time.\"\n            )\n\n        logging.info(\"Finished init ENV.\")\n\n    def init_basics(self) -> None:\n        self.dump_dir = self.configs.dump_dir\n        self.error_dir = opjoin(self.dump_dir, \"ERR\")\n        os.makedirs(self.dump_dir, exist_ok=True)\n        os.makedirs(self.error_dir, exist_ok=True)\n\n    def init_model(self) -> None:\n        self.model = RNAPro(self.configs).to(self.device)\n        num_params = sum(p.numel() for p in self.model.parameters())\n        self.print(f\"Total number of parameters: {num_params:,}\")\n\n    def load_checkpoint(self) -> None:\n        checkpoint_path = self.configs.load_checkpoint_path\n\n        if not os.path.exists(checkpoint_path):\n            raise Exception(f\"Given checkpoint path not exist [{checkpoint_path}]\")\n        self.print(\n            f\"Loading from {checkpoint_path}, strict: {self.configs.load_strict}\"\n        )\n        \n        # Handle loading checkpoints saved on different devices (XLA/CUDA/CPU)\n        # Load to CPU first, then move to target device\n        checkpoint = torch.load(checkpoint_path, map_location='cpu')\n\n        sample_key = [k for k in checkpoint[\"model\"].keys()][0]\n        # self.print(f\"Sampled key: {sample_key}\")\n        if sample_key.startswith(\"module.\"):  # DDP checkpoint has module. prefix\n            checkpoint[\"model\"] = {\n                k[len(\"module.\"):]: v for k, v in checkpoint[\"model\"].items()\n            }\n        self.model.load_state_dict(\n            state_dict=checkpoint[\"model\"],\n            strict=True,\n        )\n        self.model.eval()\n        # self.print(\"Finish loading checkpoint.\")\n\n    def init_dumper(\n        self, need_atom_confidence: bool = False, sorted_by_ranking_score: bool = True\n    ):\n        self.dumper = DataDumper(\n            base_dir=self.dump_dir,\n            need_atom_confidence=need_atom_confidence,\n            sorted_by_ranking_score=sorted_by_ranking_score,\n        )\n\n    def print_dict(self, d):\n        for k, v in d.items():\n            if isinstance(v, torch.Tensor):\n                print(f\"{k}: \", v.shape)\n            else:\n                pass\n                # print(f\"{k}: {v}\")\n\n    # Adapted from runner.train.Trainer.evaluate\n    @torch.no_grad()\n    def predict(self, data: Mapping[str, Mapping[str, Any]]) -> dict[str, torch.Tensor]:\n        eval_precision = {\n            \"fp32\": torch.float32,\n            \"bf16\": torch.bfloat16,\n            \"fp16\": torch.float16,\n        }[self.configs.dtype]\n        # print(\"eval_precision: \", eval_precision)\n        enable_amp = (\n            torch.autocast(device_type=\"cuda\", dtype=eval_precision)\n            if torch.cuda.is_available()\n            else nullcontext()\n        )\n        #         print('input_feature_dict: ', self.print_dict(data[\"input_feature_dict\"]))\n        #         exit(0)\n\n        data = to_device(data, self.device)\n        with enable_amp:\n            prediction, _, _ = self.model(\n                input_feature_dict=data[\"input_feature_dict\"],\n                label_full_dict=None,\n                label_dict=None,\n                mode=\"inference\",\n            )\n\n        return prediction\n\n    def print(self, msg: str):\n        if DIST_WRAPPER.rank == 0:\n            # logger.info(msg)\n            print(msg)\n\n    def update_model_configs(self, new_configs: Any) -> None:\n        self.model.configs = new_configs\n\n\ndef update_inference_configs(configs: Any, N_token: int):\n    # Setting the default inference configs for different N_token and N_atom\n    # when N_token is larger than 3000, the default config might OOM even on a\n    # A100 80G GPUS,\n    if N_token > 3840:\n        configs.skip_amp.confidence_head = False\n        configs.skip_amp.sample_diffusion = False\n    elif N_token > 2560:\n        configs.skip_amp.confidence_head = False\n        configs.skip_amp.sample_diffusion = True\n    else:\n        configs.skip_amp.confidence_head = True\n        configs.skip_amp.sample_diffusion = True\n    return configs\n\n\ndef infer_predict(runner: InferenceRunner, configs: Any) -> None:\n    # Data\n    # logger.info(f\"Loading data from {configs.input_json_path}\")\n    try:\n        dataloader = get_inference_dataloader(configs=configs)\n    except Exception as e:\n        error_message = f\"{e}:\\n{traceback.format_exc()}\"\n        logger.info(error_message)\n        with open(opjoin(runner.error_dir, \"error.txt\"), \"a\") as f:\n            f.write(error_message)\n        return\n\n    num_data = len(dataloader.dataset)\n    for seed in configs.seeds:\n        seed_everything(seed=seed, deterministic=configs.deterministic)\n        for batch in dataloader:\n            try:\n                data, atom_array, data_error_message = batch[0]\n                sample_name = data[\"sample_name\"]\n\n                if len(data_error_message) > 0:\n                    logger.info(data_error_message)\n                    with open(opjoin(runner.error_dir, f\"{sample_name}.txt\"), \"a\") as f:\n                        f.write(data_error_message)\n                    continue\n\n                logger.info(\n                    (\n                        f\"[Rank {DIST_WRAPPER.rank} ({data['sample_index'] + 1}/{num_data})] {sample_name}: \"\n                        f\"N_asym {data['N_asym'].item()}, N_token {data['N_token'].item()}, \"\n                        f\"N_atom {data['N_atom'].item()}, N_msa {data['N_msa'].item()}\"\n                    )\n                )\n                new_configs = update_inference_configs(configs, data[\"N_token\"].item())\n                runner.update_model_configs(new_configs)\n                prediction = runner.predict(data)\n                runner.dumper.dump(\n                    dataset_name=\"\",\n                    pdb_id=sample_name,\n                    seed=seed,\n                    pred_dict=prediction,\n                    atom_array=atom_array,\n                    entity_poly_type=data[\"entity_poly_type\"],\n                )\n\n                logger.info(\n                    f\"[Rank {DIST_WRAPPER.rank}] {data['sample_name']} succeeded - \"\n                    f\"Results saved to {configs.dump_dir}\"\n                )\n                torch.cuda.empty_cache()\n            except Exception as e:\n                error_message = f\"[Rank {DIST_WRAPPER.rank}]{data['sample_name']} {e}:\\n{traceback.format_exc()}\"\n                logger.info(error_message)\n                # Save error info\n                with open(opjoin(runner.error_dir, f\"{sample_name}.txt\"), \"a\") as f:\n                    f.write(error_message)\n                if hasattr(torch.cuda, \"empty_cache\"):\n                    torch.cuda.empty_cache()\n\n\n# data helper\ndef make_dummy_solution(valid_df):\n    solution = dotdict()\n    for i, row in valid_df.iterrows():\n        target_id = row.target_id\n        sequence = row.sequence\n        solution[target_id] = dotdict(\n            target_id=target_id,\n            sequence=sequence,\n            coord=[],\n        )\n    return solution\n\n\ndef solution_to_submit_df(solution):\n    submit_df = []\n    for k, s in solution.items():\n        df = coord_to_df(s.sequence, s.coord, s.target_id)\n        submit_df.append(df)\n\n    submit_df = pd.concat(submit_df)\n    return submit_df\n\n\ndef coord_to_df(sequence, coord, target_id):\n    L = len(sequence)\n    df = pd.DataFrame()\n    df[\"ID\"] = [f\"{target_id}_{i + 1}\" for i in range(L)]\n    df[\"resname\"] = [s for s in sequence]\n    df[\"resid\"] = [i + 1 for i in range(L)]\n\n    num_coord = len(coord)\n    for j in range(num_coord):\n        df[f\"x_{j+1}\"] = coord[j][:, 0]\n        df[f\"y_{j+1}\"] = coord[j][:, 1]\n        df[f\"z_{j+1}\"] = coord[j][:, 2]\n    return df\n\n\ndef main(configs: Any) -> None:\n    # Runner\n    runner = InferenceRunner(configs)\n    infer_predict(runner, configs)\n\n\ndef create_input_json(sequence, target_id):\n    # print(\"input_no_msa\")\n    input_json = [\n        {\n            \"sequences\": [\n                {\n                    \"rnaSequence\": {\n                        \"sequence\": sequence,\n                        \"count\": 1,\n                    }\n                }\n            ],\n            \"name\": target_id,\n        }\n    ]\n    return input_json\n\n\ndef extract_c1_coordinates(cif_file_path):\n    try:\n        # Read the CIF file using the correct biotite method\n        with open(cif_file_path, \"r\") as f:\n            cif_data = pdbx.CIFFile.read(f)\n\n        # Get structure from CIF data\n        atom_array = pdbx.get_structure(cif_data, model=1)\n\n        # Clean atom names and find C1' atoms\n        atom_names_clean = np.char.strip(atom_array.atom_name.astype(str))\n        mask_c1 = atom_names_clean == \"C1'\"\n        c1_atoms = atom_array[mask_c1]\n\n        if len(c1_atoms) == 0:\n            print(f\"Warning: No C1' atoms found in {cif_file_path}\")\n            return None\n\n        # Sort by residue ID and return coordinates\n        sort_indices = np.argsort(c1_atoms.res_id)\n        c1_atoms_sorted = c1_atoms[sort_indices]\n        c1_coords = c1_atoms_sorted.coord\n\n        return c1_coords\n    except Exception as e:\n        print(f\"Error extracting C1' coordinates from {cif_file_path}: {e}\")\n        return None\n\n\ndef process_sequence(sequence, target_id, temp_dir):\n    # Create input JSON\n    input_json = create_input_json(sequence, target_id)\n\n    # Save JSON to temporary file\n    os.makedirs(temp_dir, exist_ok=True)\n    input_json_path = os.path.join(temp_dir, f\"{target_id}_input.json\")\n    with open(input_json_path, \"w\") as f:\n        json.dump(input_json, f, indent=4)\n\n\ndef run_ptx(target_id, sequence, configs, solution, template_idx, runner):\n    # Create directories\n    temp_dir = f\"./{configs.dump_dir}/input\"  # Same as in kaggle_inference.py\n    output_dir = f\"./{configs.dump_dir}/output\"  # Same as in kaggle_inference.py\n    os.makedirs(temp_dir, exist_ok=True)\n    os.makedirs(output_dir, exist_ok=True)\n\n    process_sequence(sequence=sequence, target_id=target_id, temp_dir=temp_dir)\n    configs.input_json_path = os.path.join(temp_dir, f\"{target_id}_input.json\")\n    configs.template_idx = int(template_idx)\n\n    infer_predict(runner, configs)\n\n    cif_file_path = (\n        f\"{configs.dump_dir}/{target_id}/seed_42/predictions/{target_id}_sample_0.cif\"\n    )\n    cif_new_path = f\"{configs.dump_dir}/{target_id}/seed_42/predictions/{target_id}_sample_{template_idx}_new.cif\"\n    shutil.copy(cif_file_path, cif_new_path)\n    coord = extract_c1_coordinates(cif_file_path)\n    if coord is None:\n        coord = np.zeros((len(sequence), 3), dtype=np.float32)\n    elif coord.shape[0] < (len(sequence)):\n        pad_len = len(sequence) - coord.shape[0]\n        pad = np.zeros((pad_len, 3), dtype=np.float32)\n        coord = np.concatenate([coord, pad], axis=0)\n    solution[target_id].coord.append(coord)\n\n\ndef run() -> None:\n    LOG_FORMAT = \"%(asctime)s,%(msecs)-3d %(levelname)-8s [%(filename)s:%(lineno)s %(funcName)s] %(message)s\"\n    logging.basicConfig(\n        format=LOG_FORMAT,\n        level=logging.WARNING,\n        datefmt=\"%Y-%m-%d %H:%M:%S\",\n        filemode=\"w\",\n    )\n    # Silence dataloader and rnapro module logging\n    logging.getLogger(\"rnapro.data\").setLevel(logging.WARNING)\n    logging.getLogger(\"rnapro\").setLevel(logging.WARNING)\n    configs_base[\"use_deepspeed_evo_attention\"] = (\n        os.environ.get(\"USE_DEEPSPEED_EVO_ATTENTION\", False) == \"true\"\n    )\n    configs = {**configs_base, **{\"data\": data_configs}, **inference_configs}\n    configs = parse_configs(\n        configs=configs,\n        arg_str=parse_sys_args(),\n        fill_required_with_null=True,\n    )\n\n    valid_df = pd.read_csv(configs.sequences_csv)\n    print(f\"\\n -> Loaded {len(valid_df)} sequence(s)\")\n\n    # Build model and load checkpoint once before looping over sequences\n\n    print('\\n -> Building model and loading checkpoint')\n    runner = InferenceRunner(configs)\n    print('\\n -> Done, starting inference...')\n\n    solution = make_dummy_solution(valid_df)\n    for idx, row in valid_df.iterrows():\n        print(f\"\\n -> Sequence {row.target_id}: {row.sequence}\")\n\n        if len(row.sequence) > configs.max_len:\n            print(f'Sequence is too long ({len(row.sequence)} > {configs.max_len}), skipping')\n            for template_idx in range(configs.n_templates_inf):\n                coord = np.zeros((len(row.sequence), 3), dtype=np.float32)\n                solution[row.target_id].coord.append(coord)\n            continue\n\n        try:\n            target_id = row.target_id\n            sequence = row.sequence\n            for template_idx in range(configs.n_templates_inf):\n                print()\n                run_ptx(\n                    target_id=target_id,\n                    sequence=sequence,\n                    configs=configs,\n                    solution=solution,\n                    template_idx=template_idx,\n                    runner=runner,\n                )\n        except Exception as e:\n            print(f\"Error processing {row.target_id}: {e}\")\n            continue\n\n    print('\\n\\n -> Inference done ! Saving to submission.csv')\n    submit_df = solution_to_submit_df(solution)\n    submit_df = submit_df.fillna(0.0)\n    submit_df.to_csv(\"./submission.csv\", index=False)\n\n\nif __name__ == \"__main__\":\n    run()","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:33:40.025502Z","iopub.status.busy":"2026-02-16T19:33:40.025274Z","iopub.status.idle":"2026-02-16T19:33:40.036855Z","shell.execute_reply":"2026-02-16T19:33:40.036198Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.026324,"end_time":"2026-02-16T19:33:40.038306","exception":false,"start_time":"2026-02-16T19:33:40.011982","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"83c1a6e3","cell_type":"code","source":"%%writefile rnapro_inference_kaggle.sh\n\nexport LAYERNORM_TYPE=torch # fast_layernorm, torch\n\n\n# Inference parameters (RNAPro)\nSEED=42\nN_SAMPLE=1\nN_STEP=225\nN_CYCLE=10\n\n# Paths\nDUMP_DIR=\"../output\"\n# Set a valid checkpoint file path below\nCHECKPOINT_PATH=\"/kaggle/input/rnapro-bf16-weights/5999_ema_0.995.pt\"\n\n# Template/MSA settings\nTEMPLATE_DATA=\"./release_data/kaggle/templates.pt\"\n# Note: template_idx supports 5 choices and maps to top-k:\n# 0->top1, 1->top2, 2->top3, 3->top4, 4->top5\nTEMPLATE_IDX=0\nRNA_MSA_DIR=\"/kaggle/input/stanford-rna-3d-folding-2/MSA\"\n\nSEQUENCES_CSV=\"/kaggle/working/sample_sequences.csv\"\n\n# RibonanzaNet2 path (keep as-is per request)\nRIBONANZA_PATH=\"/kaggle/input/ribonanzanet2/pytorch/alpha/1/\"\n\nMODEL_NAME=\"rnapro_base\"\nmkdir -p \"${DUMP_DIR}\"\n\npython3 runner/inference.py \\\n    --model_name \"${MODEL_NAME}\" \\\n    --seeds ${SEED} \\\n    --dump_dir \"${DUMP_DIR}\" \\\n    --load_checkpoint_path \"${CHECKPOINT_PATH}\" \\\n    --use_msa true \\\n    --use_template \"ca_precomputed\" \\\n    --model.use_template \"ca_precomputed\" \\\n    --model.use_RibonanzaNet2 true \\\n    --model.template_embedder.n_blocks 2 \\\n    --model.ribonanza_net_path \"${RIBONANZA_PATH}\" \\\n    --template_data \"${TEMPLATE_DATA}\" \\\n    --template_idx ${TEMPLATE_IDX} \\\n    --rna_msa_dir \"${RNA_MSA_DIR}\" \\\n    --model.N_cycle ${N_CYCLE} \\\n    --sample_diffusion.N_sample ${N_SAMPLE} \\\n    --sample_diffusion.N_step ${N_STEP} \\\n    --load_strict true \\\n    --num_workers 0 \\\n    --triangle_attention \"torch\" \\\n    --triangle_multiplicative \"torch\" \\\n    --sequences_csv \"${SEQUENCES_CSV}\" \\\n    --max_len 512 \\\n    --n_templates_inf 5\n\n\n# --triangle_attention supports 'triattention', 'cuequivariance', 'deepspeed', 'torch'\n# --triangle_multiplicative supports 'cuequivariance', 'torch'\n# --max_len 1000: Sequences longer than max_len will be skipped to avoid oom\n","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:33:40.063285Z","iopub.status.busy":"2026-02-16T19:33:40.063087Z","iopub.status.idle":"2026-02-16T19:33:40.067349Z","shell.execute_reply":"2026-02-16T19:33:40.066717Z"},"papermill":{"duration":0.018312,"end_time":"2026-02-16T19:33:40.068623","exception":false,"start_time":"2026-02-16T19:33:40.050311","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"48f47c88","cell_type":"code","source":"!bash ./rnapro_inference_kaggle.sh\n","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:33:40.093535Z","iopub.status.busy":"2026-02-16T19:33:40.09332Z","iopub.status.idle":"2026-02-16T19:45:54.894268Z","shell.execute_reply":"2026-02-16T19:45:54.892942Z"},"papermill":{"duration":734.815903,"end_time":"2026-02-16T19:45:54.896687","exception":false,"start_time":"2026-02-16T19:33:40.080784","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"c36efeeb","cell_type":"code","source":"!mv /kaggle/working/RNAPro/submission.csv /kaggle/working/submission.csv\n","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:45:54.941171Z","iopub.status.busy":"2026-02-16T19:45:54.940879Z","iopub.status.idle":"2026-02-16T19:45:55.107542Z","shell.execute_reply":"2026-02-16T19:45:55.1067Z"},"papermill":{"duration":0.190628,"end_time":"2026-02-16T19:45:55.109357","exception":false,"start_time":"2026-02-16T19:45:54.918729","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"792b5a46","cell_type":"code","source":"%cd /kaggle/working\n","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:45:55.140246Z","iopub.status.busy":"2026-02-16T19:45:55.139966Z","iopub.status.idle":"2026-02-16T19:45:55.145474Z","shell.execute_reply":"2026-02-16T19:45:55.144852Z"},"papermill":{"duration":0.022281,"end_time":"2026-02-16T19:45:55.146852","exception":false,"start_time":"2026-02-16T19:45:55.124571","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"6fc78d09","cell_type":"code","source":"!head /kaggle/working/submission.csv\n","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:45:55.177383Z","iopub.status.busy":"2026-02-16T19:45:55.177164Z","iopub.status.idle":"2026-02-16T19:45:55.342662Z","shell.execute_reply":"2026-02-16T19:45:55.341972Z"},"papermill":{"duration":0.183197,"end_time":"2026-02-16T19:45:55.344424","exception":false,"start_time":"2026-02-16T19:45:55.161227","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"ebecf139","cell_type":"markdown","source":"## Self-Distillation\nRound 1 RNAPro output -> new templates -> Round 2 RNAPro\n","metadata":{"papermill":{"duration":0.01458,"end_time":"2026-02-16T19:45:55.374337","exception":false,"start_time":"2026-02-16T19:45:55.359757","status":"completed"},"tags":[]}},{"id":"5507f024","cell_type":"code","source":"%cd /kaggle/working/RNAPro\n!python preprocess/convert_templates_to_pt_files.py --input_csv /kaggle/working/submission.csv --output_name templates.pt\nprint(\"New templates generated from Round 1 RNAPro predictions.\")\n","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:45:55.405113Z","iopub.status.busy":"2026-02-16T19:45:55.404835Z","iopub.status.idle":"2026-02-16T19:45:58.305152Z","shell.execute_reply":"2026-02-16T19:45:58.304174Z"},"papermill":{"duration":2.917981,"end_time":"2026-02-16T19:45:58.306736","exception":false,"start_time":"2026-02-16T19:45:55.388755","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"85746494","cell_type":"code","source":"# Round 2 - self-distillation\n!bash ./rnapro_inference_kaggle.sh\n","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:45:58.33872Z","iopub.status.busy":"2026-02-16T19:45:58.33839Z","iopub.status.idle":"2026-02-16T19:57:41.831483Z","shell.execute_reply":"2026-02-16T19:57:41.830713Z"},"papermill":{"duration":703.511363,"end_time":"2026-02-16T19:57:41.83357","exception":false,"start_time":"2026-02-16T19:45:58.322207","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"72308edf","cell_type":"code","source":"!mv /kaggle/working/RNAPro/submission.csv /kaggle/working/submission.csv\n%cd /kaggle/working\n\nimport pandas as pd\ndf_rnapro = pd.read_csv(\"/kaggle/working/submission.csv\")\nprint(f\"RNAPro predictions: {df_rnapro.shape}\")\ndf_rnapro.to_csv(\"/kaggle/working/submission_rnapro.csv\", index=False)\nprint(\"Saved submission_rnapro.csv\")\n","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:57:41.870279Z","iopub.status.busy":"2026-02-16T19:57:41.86998Z","iopub.status.idle":"2026-02-16T19:57:42.09453Z","shell.execute_reply":"2026-02-16T19:57:42.093587Z"},"papermill":{"duration":0.244389,"end_time":"2026-02-16T19:57:42.095987","exception":false,"start_time":"2026-02-16T19:57:41.851598","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"dd37116e","cell_type":"markdown","source":"## Boltz-1 Integration\nOpen-source biomolecular structure prediction (AlphaFold3-class).\nEnsemble with RNAPro for diverse predictions.\n","metadata":{"papermill":{"duration":0.017248,"end_time":"2026-02-16T19:57:42.130984","exception":false,"start_time":"2026-02-16T19:57:42.113736","status":"completed"},"tags":[]}},{"id":"224e3e8c","cell_type":"code","source":"# Install Boltz-1 from Kaggle datasets\n# Required datasets (Add Data ile ekle):\n#   - boltz_dependencies (Youhan Lee)\n#   - rna_prediction_boltz (Youhan Lee)\n#   - fairscale-0413\nimport subprocess, os, sys\n\nboltz_installed = False\n\n# Install from Kaggle datasets (offline - works during scoring)\nBOLTZ_DEPS = \"/kaggle/input/boltz-dependencies\"\nFAIRSCALE = \"/kaggle/input/fairscale-0413\"\n\ntry:\n    import boltz\n    boltz_installed = True\n    print(f\"Boltz already installed\")\nexcept ImportError:\n    if os.path.exists(BOLTZ_DEPS):\n        print(\"Installing Boltz from Kaggle datasets...\")\n        # Install fairscale first if available\n        if os.path.exists(FAIRSCALE):\n            !pip install -q --no-index --find-links {FAIRSCALE} fairscale\n        # Install boltz + deps\n        !pip install -q --no-index --find-links {BOLTZ_DEPS} boltz\n        try:\n            import boltz\n            boltz_installed = True\n            print(\"Boltz installed from Kaggle datasets!\")\n        except Exception as e:\n            print(f\"Install failed: {e}\")\n    else:\n        # Fallback: pip install (internet required)\n        try:\n            !pip install -q boltz\n            import boltz\n            boltz_installed = True\n            print(\"Boltz installed via pip\")\n        except:\n            print(\"Could not install Boltz. Add datasets: boltz_dependencies, rna_prediction_boltz, fairscale-0413\")\n\nprint(f\"boltz_installed = {boltz_installed}\")\n","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:57:42.166352Z","iopub.status.busy":"2026-02-16T19:57:42.166076Z","iopub.status.idle":"2026-02-16T20:01:03.012498Z","shell.execute_reply":"2026-02-16T20:01:03.011585Z"},"papermill":{"duration":200.866318,"end_time":"2026-02-16T20:01:03.014368","exception":false,"start_time":"2026-02-16T19:57:42.14805","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"dc5b0dd6","cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os, time, glob, json\n\ntest_seqs = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding-2/test_sequences.csv\")\nIS_SCORING_RUN = os.environ.get(\"KAGGLE_IS_COMPETITION_RERUN\")\nif not IS_SCORING_RUN:\n    test_seqs = test_seqs.head(5)\n\nBOLTZ_OUT = \"/kaggle/working/boltz_output\"\nBOLTZ_INPUT = \"/kaggle/working/boltz_input\"\nos.makedirs(BOLTZ_INPUT, exist_ok=True)\nos.makedirs(BOLTZ_OUT, exist_ok=True)\n\nif boltz_installed:\n    print(f\"Preparing Boltz input for {len(test_seqs)} sequences...\")\n    \n    # Create YAML input files for each sequence\n    for idx, row in test_seqs.iterrows():\n        target_id = row[\"target_id\"]\n        sequence = row[\"sequence\"]\n        \n        # Skip very long sequences (>1500) to avoid OOM\n        if len(sequence) > 1500:\n            print(f\"  Skipping {target_id} (len={len(sequence)}, too long for Boltz)\")\n            continue\n        \n        yaml_content = f\"\"\"version: 1\nsequences:\n  - rna:\n      id: A\n      sequence: {sequence}\n\"\"\"\n        yaml_path = os.path.join(BOLTZ_INPUT, f\"{target_id}.yaml\")\n        with open(yaml_path, \"w\") as f:\n            f.write(yaml_content)\n    \n    yaml_files = glob.glob(os.path.join(BOLTZ_INPUT, \"*.yaml\"))\n    print(f\"Created {len(yaml_files)} YAML input files\")\nelse:\n    print(\"Boltz not installed, skipping input preparation.\")\n","metadata":{"execution":{"iopub.execute_input":"2026-02-16T20:01:03.055378Z","iopub.status.busy":"2026-02-16T20:01:03.055101Z","iopub.status.idle":"2026-02-16T20:01:03.067248Z","shell.execute_reply":"2026-02-16T20:01:03.066489Z"},"papermill":{"duration":0.035043,"end_time":"2026-02-16T20:01:03.068643","exception":false,"start_time":"2026-02-16T20:01:03.0336","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"0b578e61","cell_type":"code","source":"# Run Boltz-1 inference\nif boltz_installed:\n    import torch\n    torch.cuda.empty_cache()  # Free RNAPro GPU memory\n    \n    # Set Boltz cache to pre-downloaded weights\n    BOLTZ_CACHE = \"/kaggle/input/rna-prediction-boltz\"\n    if os.path.exists(BOLTZ_CACHE):\n        os.environ[\"BOLTZ_CACHE\"] = BOLTZ_CACHE\n        print(f\"Using cached Boltz weights from {BOLTZ_CACHE}\")\n    \n    yaml_files = sorted(glob.glob(os.path.join(BOLTZ_INPUT, \"*.yaml\")))\n    print(f\"Running Boltz on {len(yaml_files)} sequences...\")\n    \n    for i, yaml_path in enumerate(yaml_files):\n        target_id = os.path.basename(yaml_path).replace(\".yaml\", \"\")\n        print(f\"  [{i+1}/{len(yaml_files)}] {target_id}...\", end=\" \")\n        t0 = time.time()\n        \n        try:\n            result = subprocess.run(\n                [\"boltz\", \"predict\", yaml_path,\n                 \"--out_dir\", BOLTZ_OUT,\n                 \"--recycling_steps\", \"3\",\n                 \"--diffusion_samples\", \"5\",\n                 \"--output_format\", \"pdb\"],\n                capture_output=True, text=True, timeout=600\n            )\n            if result.returncode == 0:\n                print(f\"done ({time.time()-t0:.0f}s)\")\n            else:\n                print(f\"FAILED: {result.stderr[-200:]}\")\n        except subprocess.TimeoutExpired:\n            print(\"TIMEOUT\")\n        except Exception as e:\n            print(f\"ERROR: {e}\")\n    \n    print(\"Boltz inference complete!\")\nelse:\n    print(\"Boltz not installed, skipping inference.\")\n","metadata":{"execution":{"iopub.execute_input":"2026-02-16T20:01:03.107682Z","iopub.status.busy":"2026-02-16T20:01:03.107107Z","iopub.status.idle":"2026-02-16T20:01:03.11394Z","shell.execute_reply":"2026-02-16T20:01:03.11329Z"},"papermill":{"duration":0.027421,"end_time":"2026-02-16T20:01:03.115359","exception":false,"start_time":"2026-02-16T20:01:03.087938","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"7e36134e","cell_type":"code","source":"# Parse Boltz PDB output -> extract C1 prime coordinates\ndef parse_boltz_pdb(pdb_path):\n    \"\"\"Extract C1 prime atom coordinates from a PDB file.\"\"\"\n    coords = []\n    with open(pdb_path, \"r\") as f:\n        for line in f:\n            if line.startswith(\"ATOM\") or line.startswith(\"HETATM\"):\n                atom_name = line[12:16].strip()\n                if atom_name == \"C1'\":\n                    x = float(line[30:38])\n                    y = float(line[38:46])\n                    z = float(line[46:54])\n                    coords.append([x, y, z])\n    return np.array(coords) if coords else None\n\nboltz_predictions = {}  # target_id -> list of coord arrays\n\nif boltz_installed:\n    for target_dir in glob.glob(os.path.join(BOLTZ_OUT, \"predictions\", \"*\")):\n        target_id = os.path.basename(target_dir)\n        pdb_files = sorted(glob.glob(os.path.join(target_dir, \"*.pdb\")))\n        \n        if pdb_files:\n            preds = []\n            for pdb_file in pdb_files[:5]:\n                coords = parse_boltz_pdb(pdb_file)\n                if coords is not None:\n                    preds.append(coords)\n            if preds:\n                boltz_predictions[target_id] = preds\n    \n    print(f\"Parsed Boltz predictions for {len(boltz_predictions)} targets\")\n    for tid, preds in list(boltz_predictions.items())[:3]:\n        print(f\"  {tid}: {len(preds)} predictions, shape={preds[0].shape}\")\nelse:\n    print(\"Boltz not installed, no predictions to parse.\")\n","metadata":{"execution":{"iopub.execute_input":"2026-02-16T20:01:03.153229Z","iopub.status.busy":"2026-02-16T20:01:03.152718Z","iopub.status.idle":"2026-02-16T20:01:03.160251Z","shell.execute_reply":"2026-02-16T20:01:03.15948Z"},"papermill":{"duration":0.028538,"end_time":"2026-02-16T20:01:03.161688","exception":false,"start_time":"2026-02-16T20:01:03.13315","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"a996d02a","cell_type":"markdown","source":"## Ensemble: RNAPro + Boltz-1\nCombine predictions from both models for maximum diversity.\n","metadata":{"papermill":{"duration":0.018271,"end_time":"2026-02-16T20:01:03.198646","exception":false,"start_time":"2026-02-16T20:01:03.180375","status":"completed"},"tags":[]}},{"id":"46e7e726","cell_type":"code","source":"# Ensemble RNAPro + Boltz predictions\ndf_rnapro = pd.read_csv(\"/kaggle/working/submission_rnapro.csv\")\ndf_ensemble = df_rnapro.copy()\n\nif boltz_predictions:\n    # For each target with Boltz predictions, replace slots 4-5 with Boltz\n    targets_updated = 0\n    for target_id, boltz_preds in boltz_predictions.items():\n        # Find rows for this target\n        mask = df_ensemble[\"ID\"].str.startswith(target_id + \"_\")\n        if mask.sum() == 0:\n            continue\n        \n        target_rows = df_ensemble[mask].copy()\n        n_residues = len(target_rows)\n        \n        # Use Boltz predictions for slots 4-5 (keep RNAPro for 1-3)\n        for slot_idx, boltz_pred in enumerate(boltz_preds[:2]):\n            slot = slot_idx + 4  # slots 4 and 5\n            if len(boltz_pred) == n_residues:\n                df_ensemble.loc[mask, f\"x_{slot}\"] = boltz_pred[:, 0]\n                df_ensemble.loc[mask, f\"y_{slot}\"] = boltz_pred[:, 1]\n                df_ensemble.loc[mask, f\"z_{slot}\"] = boltz_pred[:, 2]\n                targets_updated += 1\n            elif len(boltz_pred) > n_residues:\n                # Boltz might predict more atoms, take first n_residues\n                df_ensemble.loc[mask, f\"x_{slot}\"] = boltz_pred[:n_residues, 0]\n                df_ensemble.loc[mask, f\"y_{slot}\"] = boltz_pred[:n_residues, 1]\n                df_ensemble.loc[mask, f\"z_{slot}\"] = boltz_pred[:n_residues, 2]\n                targets_updated += 1\n    \n    print(f\"Updated {targets_updated} target slots with Boltz predictions\")\nelse:\n    print(\"No Boltz predictions available - using RNAPro only\")\n\ndf_ensemble.to_csv(\"/kaggle/working/submission.csv\", index=False)\nprint(f\"Ensemble submission saved: {df_ensemble.shape}\")\n!head /kaggle/working/submission.csv\n","metadata":{"execution":{"iopub.execute_input":"2026-02-16T20:01:03.236818Z","iopub.status.busy":"2026-02-16T20:01:03.236481Z","iopub.status.idle":"2026-02-16T20:01:03.471751Z","shell.execute_reply":"2026-02-16T20:01:03.47078Z"},"papermill":{"duration":0.256433,"end_time":"2026-02-16T20:01:03.473571","exception":false,"start_time":"2026-02-16T20:01:03.217138","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"a3c85a04","cell_type":"code","source":"df_tbm = pd.read_csv(\"/kaggle/working/submission_long.csv\")\ndf_rnapro = pd.read_csv(\"/kaggle/working/submission.csv\")\ndf_seqs = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding-2/test_sequences.csv\")\nlong_targets = df_seqs[df_seqs['sequence'].str.len() > 512]['target_id'].values\n\nprint(f\"Targets to replace with TBM (len > 512): {long_targets}\")\n\nmask_long = df_rnapro['ID'].apply(lambda x: any(str(x).startswith(t + \"_\") for t in long_targets))\n\nif mask_long.sum() > 0:\n    print(f\"Replacing {mask_long.sum()} residues with TBM predictions...\")\n    df_rnapro_idx = df_rnapro.set_index('ID')\n    df_tbm_idx = df_tbm.set_index('ID')\n    ids_to_update = df_rnapro_idx[mask_long.values].index\n    valid_ids = [i for i in ids_to_update if i in df_tbm_idx.index]\n    df_rnapro_idx.loc[valid_ids] = df_tbm_idx.loc[valid_ids]\n    df_final = df_rnapro_idx.reset_index()\n    df_final.to_csv(\"/kaggle/working/submission.csv\", index=False)\n    print(\"Merged submission saved to submission.csv\")\nelse:\n    print(\"No long targets found to replace. Keeping RNAPro submission as is.\")\n","metadata":{"execution":{"iopub.execute_input":"2026-02-16T20:01:03.512665Z","iopub.status.busy":"2026-02-16T20:01:03.512361Z","iopub.status.idle":"2026-02-16T20:01:03.63085Z","shell.execute_reply":"2026-02-16T20:01:03.629958Z"},"papermill":{"duration":0.139972,"end_time":"2026-02-16T20:01:03.63253","exception":false,"start_time":"2026-02-16T20:01:03.492558","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"c4310cc2","cell_type":"code","source":"sub = pd.read_csv(\"/kaggle/working/submission.csv\")\nsub","metadata":{"execution":{"iopub.execute_input":"2026-02-16T20:01:03.673791Z","iopub.status.busy":"2026-02-16T20:01:03.673144Z","iopub.status.idle":"2026-02-16T20:01:03.699199Z","shell.execute_reply":"2026-02-16T20:01:03.698346Z"},"papermill":{"duration":0.046749,"end_time":"2026-02-16T20:01:03.700889","exception":false,"start_time":"2026-02-16T20:01:03.65414","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"f60ad8c1","cell_type":"markdown","source":"Evaluation  \n---","metadata":{"papermill":{"duration":0.018007,"end_time":"2026-02-16T20:01:03.739468","exception":false,"start_time":"2026-02-16T20:01:03.721461","status":"completed"},"tags":[]}},{"id":"23f71497","cell_type":"code","source":"# %%writefile /kaggle/working/metric.py\n# #!/usr/bin/env python\n# # coding: utf-8\n\n# # In[ ]:\n\n\n# import os\n# import re\n# import math\n# import pandas as pd\n# from pathlib import Path\n# import shutil\n# import sys\n# import csv\n\n# # ---------------------\n# # Helper: parse USalign output\n# # ---------------------\n# def parse_tmscore_output(output: str) -> float:\n#     matches = re.findall(r'TM-score=\\s+([\\d.]+)', output)\n#     if len(matches) < 2:\n#         raise ValueError('No TM score found in USalign output')\n#     return float(matches[1])\n\n# # ---------------------\n# # PDB writers\n# # ---------------------\n\n# def sanitize(xyz):\n#     MIN_COORD=-999.999\n#     MAX_COORD=9999.999\n#     return min(max(xyz,MIN_COORD),MAX_COORD)\n\n# def write_target_line(atom_name, atom_serial, residue_name, chain_id, residue_num,\n#                       x_coord, y_coord, z_coord, occupancy=1.0, b_factor=0.0, atom_type='P') -> str:\n#     return f'ATOM  {atom_serial:>5d}  {atom_name:4s}{residue_name:>3s} {chain_id:1s}{residue_num:>4d}    {sanitize(x_coord):>8.3f}{sanitize(y_coord):>8.3f}{sanitize(z_coord):>8.3f}{occupancy:>6.2f}{b_factor:>6.2f}           {atom_type}\\n'\n    \n# def write2pdb(df: pd.DataFrame, xyz_id: int, target_path: str) -> int:\n#     \"\"\"\n#     Write single-chain PDB (chain 'A') using row['resid'] as residue_num.\n#     Raises exceptions on invalid data.\n#     \"\"\"\n#     resolved_cnt = 0\n#     with open(target_path, 'w') as fh:\n#         for _, row in df.iterrows():\n#             x = row[f'x_{xyz_id}']\n#             y = row[f'y_{xyz_id}']\n#             z = row[f'z_{xyz_id}']\n#             if x > -1e6 and y > -1e6 and z > -1e6:\n#                 resolved_cnt += 1\n#                 resid_num = int(row['resid'])\n#                 fh.write(write_target_line(\"C1'\", resid_num, row['resname'], 'A', resid_num, x, y, z, atom_type='C'))\n#     return resolved_cnt\n\n# def write2pdb_singlechain_native(df_native: pd.DataFrame, xyz_id: int, target_path: str) -> int:\n#     \"\"\"\n#     Write native single-chain using row['resid'] as residue numbers.\n#     Assumes all required columns exist and are valid.\n#     \"\"\"\n#     df_sorted = df_native.copy()\n#     df_sorted['__resid_int'] = df_sorted['resid'].astype(int)\n#     df_sorted = df_sorted.sort_values('__resid_int').reset_index(drop=True)\n\n#     resolved_cnt = 0\n#     with open(target_path, 'w') as fh:\n#         for _, row in df_sorted.iterrows():\n#             x = row[f'x_{xyz_id}']\n#             y = row[f'y_{xyz_id}']\n#             z = row[f'z_{xyz_id}']\n#             if x > -1e6 and y > -1e6 and z > -1e6:\n#                 resolved_cnt += 1\n#                 resid_num = int(row['resid'])\n#                 fh.write(write_target_line(\"C1'\", resid_num, row['resname'], 'A', resid_num, x, y, z, atom_type='C'))\n#     return resolved_cnt\n\n# def write2pdb_multichain_from_solution(df_solution: pd.DataFrame, xyz_id: int, target_path: str) -> int:\n#     \"\"\"\n#     Write multi-chain PDB for native solution using columns 'chain' and 'copy' to assign chain letters.\n#     Expects 'resid' convertible to int and chain/copy present. No fallbacks.\n#     \"\"\"\n#     df_sorted = df_solution.copy()\n#     df_sorted['__resid_int'] = df_sorted['resid'].astype(int)\n#     df_sorted = df_sorted.sort_values('__resid_int')\n\n#     chain_map = {}\n#     next_ord = ord('A')\n#     written = 0\n#     with open(target_path, 'w') as fh:\n#         for _, row in df_sorted.iterrows():\n#             x = row[f'x_{xyz_id}']\n#             y = row[f'y_{xyz_id}']\n#             z = row[f'z_{xyz_id}']\n#             if not (x > -1e6 and y > -1e6 and z > -1e6):\n#                 continue\n#             chain_val = row['chain']\n#             copy_key = int(row['copy'])\n#             g = (str(chain_val), copy_key)\n#             if g not in chain_map:\n#                 if next_ord <= ord('Z'):\n#                     ch = chr(next_ord)\n#                 else:\n#                     ov = next_ord - ord('Z') - 1\n#                     if ov < 26:\n#                         ch = chr(ord('a') + ov)\n#                     else:\n#                         ch = chr(ord('0') + (ov - 26) % 10)\n#                 chain_map[g] = ch\n#                 next_ord += 1\n#             chain_id = chain_map[g]\n#             written += 1\n#             resid_num = int(row['resid'])\n#             fh.write(write_target_line(\"C1'\", resid_num, row['resname'], chain_id, resid_num, x, y, z, atom_type='C'))\n#     return written\n\n# def write2pdb_multichain_from_groups(df_pred: pd.DataFrame, xyz_id: int, target_path: str, groups_list) -> (int, list):\n#     \"\"\"\n#     Write predicted multichain PDB based on a positional groups_list (tuple per residue: (chain, copy)).\n#     Requires groups_list length == number of residues in df_pred (after sorting).\n#     Returns (written_count, chain_letters_per_res).\n#     \"\"\"\n#     df_sorted = df_pred.copy()\n#     df_sorted['__resid_int'] = df_sorted['resid'].astype(int)\n#     df_sorted = df_sorted.sort_values('__resid_int').reset_index(drop=True)\n\n#     if groups_list is None or len(groups_list) != len(df_sorted):\n#         raise ValueError(\"groups_list must be provided and match number of residues in predicted df\")\n\n#     chain_map = {}\n#     next_ord = ord('A')\n#     chain_letters = []\n#     written = 0\n#     with open(target_path, 'w') as fh:\n#         for idx, row in df_sorted.iterrows():\n#             g = groups_list[idx]\n#             if isinstance(g, tuple):\n#                 gkey = (str(g[0]), int(g[1]))\n#             else:\n#                 gkey = (str(g), None)\n#             if gkey not in chain_map:\n#                 if next_ord <= ord('Z'):\n#                     ch = chr(next_ord)\n#                 else:\n#                     ov = next_ord - ord('Z') - 1\n#                     if ov < 26:\n#                         ch = chr(ord('a') + ov)\n#                     else:\n#                         ch = chr(ord('0') + (ov - 26) % 10)\n#                 chain_map[gkey] = ch\n#                 next_ord += 1\n#             chain_id = chain_map[gkey]\n#             chain_letters.append(chain_id)\n#             x = row[f'x_{xyz_id}']\n#             y = row[f'y_{xyz_id}']\n#             z = row[f'z_{xyz_id}']\n#             if x > -1e6 and y > -1e6 and z > -1e6:\n#                 written += 1\n#                 resid_num = int(row['resid'])\n#                 fh.write(write_target_line(\"C1'\", resid_num, row['resname'], chain_id, resid_num, x, y, z, atom_type='C'))\n#     return written, chain_letters\n\n# def write2pdb_singlechain_permuted_pred(df_pred: pd.DataFrame, xyz_id: int, permuted_indices: list, target_path: str) -> int:\n#     \"\"\"\n#     Create single-chain PDB by concatenating predicted residues in permuted_indices order.\n#     Output residue numbers are sequential starting at 1 and increase for every permuted position.\n#     Raises exception if indices out of range.\n#     \"\"\"\n#     df_sorted = df_pred.copy()\n#     df_sorted['__resid_int'] = df_sorted['resid'].astype(int)\n#     df_sorted = df_sorted.sort_values('__resid_int').reset_index(drop=True)\n\n#     written = 0\n#     next_res = 1\n#     with open(target_path, 'w') as fh:\n#         for idx in permuted_indices:\n#             if idx < 0 or idx >= len(df_sorted):\n#                 # strict behavior: raise error for invalid index\n#                 raise IndexError(f\"permuted index {idx} out of range for predicted residues\")\n#             row = df_sorted.iloc[idx]\n#             x = row[f'x_{xyz_id}']\n#             y = row[f'y_{xyz_id}']\n#             z = row[f'z_{xyz_id}']\n#             out_resnum = next_res\n#             if x > -1e6 and y > -1e6 and z > -1e6:\n#                 written += 1\n#                 fh.write(write_target_line(\"C1'\", out_resnum, row['resname'], 'A', out_resnum, x, y, z, atom_type='C'))\n#             next_res += 1\n#     return written\n\n# # ---------------------\n# # USalign wrappers\n# # ---------------------\n# def run_usalign_raw(predicted_pdb: str, native_pdb: str, usalign_bin='USalign', align_sequence=False, tmscore=None) -> str:\n#     cmd = f'{usalign_bin} {predicted_pdb} {native_pdb} -atom \" C1\\'\"'\n#     if tmscore is not None:\n#         cmd += f' -TMscore {tmscore}'\n#         if int(tmscore) == 0:\n#             cmd += ' -mm 1 -ter 0'\n#     elif not align_sequence:\n#         cmd += ' -TMscore 1'\n#     return os.popen(cmd).read()\n\n# def parse_usalign_chain_orders(output: str):\n#     \"\"\"\n#     Parse USalign output for both Structure_1 and Structure_2 chain lists.\n#     Returns (chain_list_structure1, chain_list_structure2).\n#     Raises if parsing fails to find either line.\n#     \"\"\"\n#     chain1 = None\n#     chain2 = None\n#     for line in output.splitlines():\n#         line = line.strip()\n#         if line.startswith('Name of Structure_1:'):\n#             parts = line.split(':')\n#             clist = []\n#             for part in parts[2:]:\n#                 token = part.strip()\n#                 if token == '':\n#                     continue\n#                 token0 = token.split()[0]\n#                 last = token0.split(',')[-1]\n#                 ch = re.sub(r'[^A-Za-z0-9]', '', last)\n#                 if ch:\n#                     clist.append(ch)\n#             chain1 = clist\n#         elif line.startswith('Name of Structure_2:'):\n#             parts = line.split(':')\n#             clist = []\n#             for part in parts[2:]:\n#                 token = part.strip()\n#                 if token == '':\n#                     continue\n#                 token0 = token.split()[0]\n#                 last = token0.split(',')[-1]\n#                 ch = re.sub(r'[^A-Za-z0-9]', '', last)\n#                 if ch:\n#                     clist.append(ch)\n#             chain2 = clist\n#     if chain1 is None or chain2 is None:\n#         raise ValueError(\"Failed to parse chain orders from USalign output\")\n#     return chain1, chain2\n\n# # ---------------------\n# # Main scoring function (no try/except, no fallbacks)\n# # ---------------------\n# def score(solution: pd.DataFrame, submission: pd.DataFrame, row_id_column_name: str, usalign_bin_hint: str = None) -> float:\n#     \"\"\"\n#     Enhanced scoring with chain-permutation handling for multicopy targets.\n#     This version contains no try/except blocks and will raise on any error.\n#     \"\"\"\n#     # determine usalign binary\n#     if usalign_bin_hint:\n#         usalign_bin = usalign_bin_hint\n#     else:\n#         if os.path.exists('/kaggle/input/usalign/USalign') and not os.path.exists('/kaggle/working/USalign'):\n#             shutil.copy2('/kaggle/input/usalign/USalign', '/kaggle/working/USalign')\n#             os.chmod('/kaggle/working/USalign', 0o755)\n#         usalign_bin = '/kaggle/working/USalign' if os.path.exists('/kaggle/working/USalign') else 'USalign'\n\n#     sol = solution.copy()\n#     sub = submission.copy()\n#     sol['target_id'] = sol['ID'].apply(lambda x: '_'.join(str(x).split('_')[:-1]))\n#     sub['target_id'] = sub['ID'].apply(lambda x: '_'.join(str(x).split('_')[:-1]))\n\n#     results = []\n\n#     for target_id, group_native in sol.groupby('target_id'):\n#         group_predicted = sub[sub['target_id'] == target_id]\n#         has_chain_copy = ('chain' in group_native.columns) and ('copy' in group_native.columns)\n#         is_multicopy = has_chain_copy and (group_native['copy'].astype(float).max() > 1)\n\n#         # precompute native models that have coords\n#         native_with_coords = []\n#         for native_cnt in range(1, 41):\n#             native_pdb = f'native_{target_id}_{native_cnt}.pdb'\n#             resolved_native = write2pdb(group_native, native_cnt, native_pdb)\n#             if resolved_native > 0:\n#                 native_with_coords.append(native_cnt)\n#             else:\n#                 if os.path.exists(native_pdb):\n#                     os.remove(native_pdb)\n\n#         if not native_with_coords:\n#             raise ValueError(f\"No native models with coordinates for target {target_id}\")\n\n#         best_per_pred = []\n#         for pred_cnt in range(1, 6):\n#             if not is_multicopy:\n#                 predicted_pdb = f'predicted_{target_id}_{pred_cnt}.pdb'\n#                 resolved_pred = write2pdb(group_predicted, pred_cnt, predicted_pdb)\n#                 if resolved_pred <= 2:\n#                     #print(f\"Predicted model {pred_cnt} for target {target_id} has insufficient coordinates\")\n#                     best_per_pred.append( 0.0 )\n#                     continue\n                \n#                 scores = []\n#                 for native_cnt in native_with_coords:\n#                     native_pdb = f'native_{target_id}_{native_cnt}.pdb'\n#                     out = run_usalign_raw(predicted_pdb, native_pdb, usalign_bin=usalign_bin, align_sequence=False, tmscore=1)\n#                     s = parse_tmscore_output(out)\n#                     scores.append(s)\n#                 best_per_pred.append(max(scores))\n\n#             else:\n#                 # multicopy\n#                 # strict: require chain and copy columns convertible\n#                 gn_sorted = group_native.copy()\n#                 gn_sorted['__resid_int'] = gn_sorted['resid'].astype(int)\n#                 gn_sorted = gn_sorted.sort_values('__resid_int').reset_index(drop=True)\n#                 groups_list = []\n#                 for _, r in gn_sorted.iterrows():\n#                     chain_val = r['chain']\n#                     copy_i = int(r['copy'])\n#                     groups_list.append((chain_val, copy_i))\n\n#                 # predicted multichain - groups_list must match predicted residue count or error\n#                 dfp_sorted = group_predicted.copy()\n#                 dfp_sorted['__resid_int'] = dfp_sorted['resid'].astype(int)\n#                 dfp_sorted = dfp_sorted.sort_values('__resid_int').reset_index(drop=True)\n#                 if len(groups_list) != len(dfp_sorted):\n#                     raise ValueError(f\"groups_list length ({len(groups_list)}) does not match predicted residue count ({len(dfp_sorted)}) for target {target_id}\")\n\n#                 predicted_multi_pdb = f'pred_multi_{target_id}_{pred_cnt}.pdb'\n#                 resolved_pred_multi, pred_chain_letters = write2pdb_multichain_from_groups(group_predicted, pred_cnt, predicted_multi_pdb, groups_list)\n#                 if resolved_pred_multi == 0:\n#                     #print(f\"Predicted multi model {pred_cnt} for target {target_id} has no coordinates\")\n#                     best_per_pred.append( 0.0 )\n#                     continue\n\n#                 scores = []\n#                 for native_cnt in native_with_coords:\n#                     native_multi_pdb = f'native_multi_{target_id}_{native_cnt}.pdb'\n#                     resolved_native_multi = write2pdb_multichain_from_solution(group_native, native_cnt, native_multi_pdb)\n#                     if resolved_native_multi == 0:\n#                         continue\n\n#                     raw_out = run_usalign_raw(predicted_multi_pdb, native_multi_pdb, usalign_bin=usalign_bin, align_sequence=True, tmscore=0)\n#                     chain1, chain2 = parse_usalign_chain_orders(raw_out)  # will raise if parsing fails\n\n#                     # build native->pred mapping chain2[i] -> chain1[i]\n#                     native_to_pred = {n_ch: p_ch for n_ch, p_ch in zip(chain2, chain1)}\n\n#                     # canonical native order = chain2 unique in order seen\n#                     #native_chain_order = []\n#                     #for ch in chain2:\n#                     #    if ch not in native_chain_order:\n#                     #        native_chain_order.append(ch)\n#                     native_chain_order = list(native_to_pred.keys())\n#                     native_chain_order.sort() # this is critical...\n\n#                     # predicted chain order by following native chain A,B,...\n#                     pred_chain_order = [native_to_pred[n_ch] for n_ch in native_chain_order if native_to_pred.get(n_ch) is not None]\n\n#                     # construct pred_positions_by_chain\n#                     pred_positions_by_chain = {}\n#                     for idx, ch in enumerate(pred_chain_letters):\n#                         if ch is None:\n#                             continue\n#                         pred_positions_by_chain.setdefault(ch, []).append(idx)\n\n#                     # require that each chain in pred_chain_order exists in pred_positions_by_chain\n#                     pred_chain_order = [p for p in pred_chain_order if p in pred_positions_by_chain]\n\n#                     # form permuted indices by concatenation\n#                     permuted_indices = []\n#                     for ch in pred_chain_order:\n#                         permuted_indices.extend(pred_positions_by_chain[ch])\n#                     # append any remaining\n#                     for idx in range(len(pred_chain_letters)):\n#                         if idx not in permuted_indices:\n#                             permuted_indices.append(idx)\n\n#                     # write permuted single-chain predicted and native single-chain\n#                     pred_single_perm = f'pred_permuted_{target_id}_{pred_cnt}_{native_cnt}.pdb'\n#                     written_pred_single = write2pdb_singlechain_permuted_pred(group_predicted, pred_cnt, permuted_indices, pred_single_perm)\n#                     native_single = f'native_single_{target_id}_{native_cnt}.pdb'\n#                     written_native = write2pdb_singlechain_native(group_native, native_cnt, native_single)\n\n#                     if written_pred_single <= 2 or written_native <= 2:\n#                         raise ValueError(f\"Insufficient residues after permutation for target {target_id}, pred {pred_cnt}, native {native_cnt}\")\n\n#                     out = run_usalign_raw(pred_single_perm, native_single, usalign_bin=usalign_bin, align_sequence=False, tmscore=1)\n#                     score_final = parse_tmscore_output(out)\n#                     scores.append(score_final)\n\n#                 best_per_pred.append(max(scores))\n\n#         results.append(max(best_per_pred))\n\n#     if not results:\n#         pass\n#         #raise ValueError(\"No targets scored\")\n#     return float(sum(results) / len(results)) if len(results)>0 else 0.0","metadata":{"execution":{"iopub.execute_input":"2026-02-16T20:01:03.777482Z","iopub.status.busy":"2026-02-16T20:01:03.777226Z","iopub.status.idle":"2026-02-16T20:01:03.788312Z","shell.execute_reply":"2026-02-16T20:01:03.787685Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.032217,"end_time":"2026-02-16T20:01:03.789752","exception":false,"start_time":"2026-02-16T20:01:03.757535","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"7483d7d4","cell_type":"code","source":"# import runpy\n# module_globals = runpy.run_path(\"/kaggle/working/metric.py\")\n# score = module_globals['score'] ","metadata":{"execution":{"iopub.execute_input":"2026-02-16T20:01:03.828756Z","iopub.status.busy":"2026-02-16T20:01:03.82843Z","iopub.status.idle":"2026-02-16T20:01:03.831735Z","shell.execute_reply":"2026-02-16T20:01:03.831062Z"},"papermill":{"duration":0.024875,"end_time":"2026-02-16T20:01:03.833063","exception":false,"start_time":"2026-02-16T20:01:03.808188","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"59818e06","cell_type":"code","source":"# if not IS_SCORING_RUN:\n#     import pandas as pd\n#     sub = pd.read_csv('/kaggle/working/submission.csv')\n#     sol = pd.read_csv('/kaggle/input/stanford-rna-3d-folding-2/validation_labels.csv')\n\n#     sub['target_id'] = sub['ID'].apply(lambda x: '_'.join(str(x).split('_')[:-1]))\n#     sol['target_id'] = sol['ID'].apply(lambda x: '_'.join(str(x).split('_')[:-1]))\n    \n#     # Get unique targets from submission\n#     sub_targets = sub['target_id'].unique()\n    \n#     results = []\n#     for target_id in sub_targets:\n#         group_native = sol[sol['target_id'] == target_id]\n#         group_predicted = sub[sub['target_id'] == target_id]\n#         result = score(group_native, group_predicted, 'ID')\n#         print(f\"{target_id}: {result:.4f} (length = {len(group_native)})\")\n#         results.append(result)\n    \n#     print(f\"\\nMean score: {sum(results)/len(results):.4f} (n={len(results)})\")","metadata":{"execution":{"iopub.execute_input":"2026-02-16T20:01:03.871916Z","iopub.status.busy":"2026-02-16T20:01:03.87162Z","iopub.status.idle":"2026-02-16T20:01:03.875382Z","shell.execute_reply":"2026-02-16T20:01:03.874611Z"},"papermill":{"duration":0.024917,"end_time":"2026-02-16T20:01:03.876945","exception":false,"start_time":"2026-02-16T20:01:03.852028","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"1f765187","cell_type":"code","source":"","metadata":{"papermill":{"duration":0.018364,"end_time":"2026-02-16T20:01:03.913878","exception":false,"start_time":"2026-02-16T20:01:03.895514","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}