{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":15231210},{"sourceType":"datasetVersion","sourceId":14604295,"datasetId":9328538,"databundleVersionId":15440074},{"sourceType":"datasetVersion","sourceId":11472091,"datasetId":7189531,"databundleVersionId":11916011},{"sourceType":"datasetVersion","sourceId":11769888,"datasetId":7389227,"databundleVersionId":12254689},{"sourceType":"datasetVersion","sourceId":11769898,"datasetId":7388010,"databundleVersionId":12254700},{"sourceType":"datasetVersion","sourceId":14874339,"datasetId":9502242,"databundleVersionId":15736806},{"sourceType":"datasetVersion","sourceId":11871860,"datasetId":7403946,"databundleVersionId":12372868},{"sourceType":"datasetVersion","sourceId":11988454,"datasetId":7540404,"databundleVersionId":12503961},{"sourceType":"datasetVersion","sourceId":10855324,"datasetId":6742586,"databundleVersionId":11219268}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, glob, subprocess, shutil, sys\nBIO_WHEEL = \"/kaggle/input/datasets/kami1976/biopython-cp312/biopython-1.86-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl\"\n\nbio_files = glob.glob(BIO_WHEEL)\n\nif bio_files:\n    subprocess.run([\n        sys.executable, \"-m\", \"pip\", \"install\", \"--no-deps\", bio_files[0]\n    ])\nelse:\n    print(\"⚠️ Biopython cp311 wheel not found — skipping\")\n\n\nfrom Bio.Align import PairwiseAligner\nprint(\"Biopython OK ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:53:40.820143Z","iopub.execute_input":"2026-02-25T14:53:40.820385Z","iopub.status.idle":"2026-02-25T14:53:45.192614Z","shell.execute_reply.started":"2026-02-25T14:53:40.820355Z","shell.execute_reply":"2026-02-25T14:53:45.191828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, glob, subprocess, shutil, sys\nBIO_WHEEL = \"/kaggle/input/datasets/tobimichigan/biotite-1-2/biotite-1.2.0-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\"\n\nbio_files = glob.glob(BIO_WHEEL)\n\nif bio_files:\n    subprocess.run([\n        sys.executable, \"-m\", \"pip\", \"install\", \"--no-deps\", bio_files[0]\n    ])\nelse:\n    print(\"⚠️ Biotite cp311 wheel not found — skipping\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:53:45.194542Z","iopub.execute_input":"2026-02-25T14:53:45.194831Z","iopub.status.idle":"2026-02-25T14:53:48.413591Z","shell.execute_reply.started":"2026-02-25T14:53:45.194795Z","shell.execute_reply":"2026-02-25T14:53:48.412888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ndef show_tree(path, max_depth=3):\n    path = os.path.abspath(path)\n    for root, dirs, files in os.walk(path):\n        depth = root[len(path):].count(os.sep)\n        if depth > max_depth:\n            continue\n        indent = \"  \" * depth\n        print(f\"{indent}{os.path.basename(root)}/\")\n        for f in files[:10]:  # limit to avoid spam\n            print(f\"{indent}  {f}\")\n\npaths = [\n    \"/kaggle/input/datasets/odat1248/boltz-0511\",\n    \"/kaggle/input/datasets/odat1248/chai1-set\",\n    \"/kaggle/input/datasets/odat1248/rna2025-runner\",\n    \"/kaggle/input/datasets/qiweiyin/protenix-v1-adjusted\",\n    \"/kaggle/input/competitions/stanford-rna-3d-folding-2\"\n]\n\nfor p in paths:\n    print(\"\\n\" + \"=\"*60)\n    print(\"PATH:\", p)\n    print(\"=\"*60)\n    show_tree(p, max_depth=3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:53:48.414628Z","iopub.execute_input":"2026-02-25T14:53:48.414974Z","iopub.status.idle":"2026-02-25T14:53:57.221365Z","shell.execute_reply.started":"2026-02-25T14:53:48.414943Z","shell.execute_reply":"2026-02-25T14:53:57.220623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nDATA_DIR = \"/kaggle/input/competitions/stanford-rna-3d-folding-2\"\n\nval_seq = pd.read_csv(DATA_DIR + \"/validation_sequences.csv\")\nval_labels = pd.read_csv(DATA_DIR + \"/validation_labels.csv\")\n\nprint(val_seq.head())\nprint(val_labels.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:53:57.222185Z","iopub.execute_input":"2026-02-25T14:53:57.222407Z","iopub.status.idle":"2026-02-25T14:53:57.789873Z","shell.execute_reply.started":"2026-02-25T14:53:57.222386Z","shell.execute_reply":"2026-02-25T14:53:57.789207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nsys.setrecursionlimit(100000)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:53:57.790753Z","iopub.execute_input":"2026-02-25T14:53:57.791053Z","iopub.status.idle":"2026-02-25T14:53:57.794552Z","shell.execute_reply.started":"2026-02-25T14:53:57.791022Z","shell.execute_reply":"2026-02-25T14:53:57.793834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil, os\n\nSRC = \"/kaggle/input/datasets/qiweiyin/protenix-v1-adjusted/Protenix-v1-adjust/Protenix-v1\"\nDST = \"/kaggle/working/protenix\"\n\nif not os.path.exists(DST):\n    shutil.copytree(SRC, DST)\n\nPROTENIX_ROOT = DST","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:53:57.795669Z","iopub.execute_input":"2026-02-25T14:53:57.796062Z","iopub.status.idle":"2026-02-25T14:54:55.819443Z","shell.execute_reply.started":"2026-02-25T14:53:57.796029Z","shell.execute_reply":"2026-02-25T14:54:55.818813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls -R /kaggle/working/protenix | head -n 50","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:54:55.821649Z","iopub.execute_input":"2026-02-25T14:54:55.821899Z","iopub.status.idle":"2026-02-25T14:54:55.955214Z","shell.execute_reply.started":"2026-02-25T14:54:55.821876Z","shell.execute_reply":"2026-02-25T14:54:55.95447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# FIX USALIGN PERMISSION\n# =========================\n\nimport shutil, os\n\nUSALIGN_SRC = \"/kaggle/input/datasets/metric/usalign/USalign\"\nUSALIGN_BIN = \"/kaggle/working/USalign\"\n\nshutil.copy(USALIGN_SRC, USALIGN_BIN)\nos.chmod(USALIGN_BIN, 0o755)\n\nprint(\"USalign ready:\", os.path.exists(USALIGN_BIN))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:54:55.956609Z","iopub.execute_input":"2026-02-25T14:54:55.957254Z","iopub.status.idle":"2026-02-25T14:54:55.997458Z","shell.execute_reply.started":"2026-02-25T14:54:55.957224Z","shell.execute_reply":"2026-02-25T14:54:55.99687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\nimport subprocess\nimport sys\n\nSAFE_BLOCKLIST = [\n    \"torch\",\n    \"nvidia\",\n    \"cuda\",\n    \"cudnn\",\n    \"cublas\",\n    \"triton\",\n    \"cu12\",\n    \"cupy\",\n    \"lightning\",\n]\n\ndef is_safe(fname, py_version):\n\n    fname_low = fname.lower()\n\n    # block dangerous packages\n    if any(x in fname_low for x in SAFE_BLOCKLIST):\n        return False\n\n    # universal wheels\n    if \"py3-none-any\" in fname_low:\n        return True\n\n    # python version match\n    if py_version in fname_low:\n        return True\n\n    return False\n\n\ndef install_safe_wheels(wheel_dir):\n\n    py_version = f\"cp{sys.version_info.major}{sys.version_info.minor}\"\n\n    wheels = glob.glob(os.path.join(wheel_dir, \"*.whl\"))\n\n    print(\"Python version:\", py_version)\n    print(\"Found wheels:\", len(wheels))\n\n    for whl in wheels:\n\n        fname = os.path.basename(whl)\n\n        if not is_safe(fname, py_version):\n            print(\"⏭️ Skipping (unsafe):\", fname)\n            continue\n\n        print(\"✅ Installing:\", fname)\n\n        subprocess.run(\n            [sys.executable, \"-m\", \"pip\", \"install\", \"--no-deps\", whl],\n            check=False\n       )\n\n\n# Path\nBOLTZ_WHEEL_DIR = \"/kaggle/input/datasets/odat1248/boltz-0511/boltz_wheel\"\n\ninstall_safe_wheels(BOLTZ_WHEEL_DIR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:54:55.998472Z","iopub.execute_input":"2026-02-25T14:54:55.998757Z","iopub.status.idle":"2026-02-25T14:56:35.024617Z","shell.execute_reply.started":"2026-02-25T14:54:55.998725Z","shell.execute_reply":"2026-02-25T14:56:35.023819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nDATA_DIR = \"/kaggle/input/competitions/stanford-rna-3d-folding-2\"\n\nBOLTZ_RUNNER = \"/kaggle/input/datasets/odat1248/rna2025-runner/boltz_runner.py\"\nCHAI_RUNNER  = \"/kaggle/input/datasets/odat1248/rna2025-runner/chai1_runner.py\"\n\nBOLTZ_CACHE = \"/kaggle/input/datasets/odat1248/boltz-0511/boltz_data\"\nCHAI_WEIGHTS = \"/kaggle/input/datasets/odat1248/chai1-set/chai1_weights\"\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:56:35.025705Z","iopub.execute_input":"2026-02-25T14:56:35.026087Z","iopub.status.idle":"2026-02-25T14:56:35.029863Z","shell.execute_reply.started":"2026-02-25T14:56:35.026063Z","shell.execute_reply":"2026-02-25T14:56:35.029223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ihm \n!pip install modelcif rdkit","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:56:35.030903Z","iopub.execute_input":"2026-02-25T14:56:35.031144Z","iopub.status.idle":"2026-02-25T14:56:52.792719Z","shell.execute_reply.started":"2026-02-25T14:56:35.031123Z","shell.execute_reply":"2026-02-25T14:56:52.792021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# FIND MSA FILE\n# =========================\n\nMSA_DIR = \"/kaggle/input/competitions/stanford-rna-3d-folding-2/MSA\"\n\ndef find_msa(target_name):\n\n    import os\n    import glob\n\n    if not os.path.exists(MSA_DIR):\n        return None\n\n    patterns = [\n        f\"{MSA_DIR}/{target_name}.MSA.fasta\",\n        f\"{MSA_DIR}/{target_name}*.fasta\",\n        f\"{MSA_DIR}/{target_name}*.a3m\",\n    ]\n\n    for p in patterns:\n        files = glob.glob(p)\n        if len(files) > 0:\n            return files[0]\n\n    return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:56:52.793939Z","iopub.execute_input":"2026-02-25T14:56:52.794292Z","iopub.status.idle":"2026-02-25T14:56:52.799603Z","shell.execute_reply.started":"2026-02-25T14:56:52.794261Z","shell.execute_reply":"2026-02-25T14:56:52.798856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# PROTENIX API INITIALIZE\n# =========================\n\nimport sys\nfrom pathlib import Path\nimport json\nimport torch\nimport numpy as np\n\nPROT_CODE = \"/kaggle/working/protenix\"\n\nsys.path.append(PROT_CODE)\nos.environ[\"PROTENIX_ROOT_DIR\"] = PROT_CODE\n\n# disable triton kernels (critical for Kaggle GPUs)\nos.environ[\"LAYERNORM_TYPE\"] = \"torch\"\n\nfrom configs.configs_base import configs as configs_base\nfrom configs.configs_data import data_configs\nfrom configs.configs_inference import inference_configs\nfrom configs.configs_model_type import model_configs\nfrom protenix.config.config import parse_configs\n\nfrom runner.inference import (\n    InferenceRunner,\n    update_gpu_compatible_configs,\n    update_inference_configs\n)\n\nfrom protenix.data.inference.infer_dataloader import InferenceDataset\n\nMODEL_NAME = \"protenix_base_default_v1.0.0\"\n\n\ndef build_protenix_config(input_json, dump_dir, seed=101):\n\n    base = {**configs_base, **{\"data\": data_configs}, **inference_configs}\n\n    def deep_update(t, p):\n        for k, v in p.items():\n            if isinstance(v, dict) and k in t and isinstance(t[k], dict):\n                deep_update(t[k], v)\n            else:\n                t[k] = v\n\n    deep_update(base, model_configs[MODEL_NAME])\n\n    arg_list = [\n        f\"--model_name {MODEL_NAME}\",\n        f\"--input_json_path {input_json}\",\n        f\"--dump_dir {dump_dir}\",\n        \"--use_msa True\",\n        \"--use_template True\",\n        \"--use_rna_msa True\",\n        f\"--seeds {seed}\"\n    ]\n\n    arg_str = \" \".join(arg_list)\n\n    cfg = parse_configs(\n        configs=base,\n        arg_str=arg_str,\n        fill_required_with_null=True\n    )\n\n    return cfg\n\n\nprint(\"Protenix API ready ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:56:52.800692Z","iopub.execute_input":"2026-02-25T14:56:52.801076Z","iopub.status.idle":"2026-02-25T14:57:00.860763Z","shell.execute_reply.started":"2026-02-25T14:56:52.801045Z","shell.execute_reply":"2026-02-25T14:57:00.860001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n\n# ---------- PROTENIX OFFLINE FIX ----------\nos.environ[\"PROTENIX_HOME\"] = PROTENIX_ROOT\nos.environ[\"HOME\"] = \"/kaggle/working\"\nos.environ[\"HF_HOME\"] = \"/kaggle/working\"\nos.environ[\"TRANSFORMERS_OFFLINE\"] = \"1\"\nos.environ[\"HF_DATASETS_OFFLINE\"] = \"1\"\n\n# IMPORTANT — disable downloads\nos.environ[\"NO_PROXY\"] = \"*\"\nos.environ[\"http_proxy\"] = \"\"\nos.environ[\"https_proxy\"] = \"\"\n\nprint(\"Offline env set ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:00.861795Z","iopub.execute_input":"2026-02-25T14:57:00.862256Z","iopub.status.idle":"2026-02-25T14:57:00.867314Z","shell.execute_reply.started":"2026-02-25T14:57:00.862229Z","shell.execute_reply":"2026-02-25T14:57:00.866523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nDATA_DIR = \"/kaggle/input/competitions/stanford-rna-3d-folding-2\"\n\nBOLTZ_RUNNER = \"/kaggle/input/datasets/odat1248/rna2025-runner/boltz_runner.py\"\nCHAI_RUNNER  = \"/kaggle/input/datasets/odat1248/rna2025-runner/chai1_runner.py\"\n\nBOLTZ_CACHE = \"/kaggle/input/datasets/odat1248/boltz-0511/boltz_data\"\nCHAI_WEIGHTS = \"/kaggle/input/datasets/odat1248/chai1-set/chai1_weights\"\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:00.868231Z","iopub.execute_input":"2026-02-25T14:57:00.868522Z","iopub.status.idle":"2026-02-25T14:57:00.996034Z","shell.execute_reply.started":"2026-02-25T14:57:00.868484Z","shell.execute_reply":"2026-02-25T14:57:00.995359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def boltz_confidence(coords):\n\n    if coords is None:\n        return -1e9\n\n    valid = np.sum(~np.isnan(coords[:,0]))\n    coverage = valid / len(coords)\n\n    diffs = np.linalg.norm(np.diff(coords, axis=0), axis=1)\n\n    if len(diffs) == 0:\n        return -1e9\n\n    smooth = np.std(diffs)\n\n    return coverage * 2.0 - smooth * 0.3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:00.997095Z","iopub.execute_input":"2026-02-25T14:57:00.997726Z","iopub.status.idle":"2026-02-25T14:57:01.010069Z","shell.execute_reply.started":"2026-02-25T14:57:00.997702Z","shell.execute_reply":"2026-02-25T14:57:01.009443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def mutate_sequence(seq, rate=0.01):\n\n    import random\n\n    bases = [\"A\", \"C\", \"G\", \"U\"]\n\n    out = list(seq)\n\n    for i in range(len(out)):\n        if random.random() < rate:\n            out[i] = random.choice(bases)\n\n    return \"\".join(out)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:01.010932Z","iopub.execute_input":"2026-02-25T14:57:01.01116Z","iopub.status.idle":"2026-02-25T14:57:01.024898Z","shell.execute_reply.started":"2026-02-25T14:57:01.011127Z","shell.execute_reply":"2026-02-25T14:57:01.024139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import subprocess\nimport tempfile\n\ndef run_boltz(sequence, name):\n\n    import subprocess, os, shutil, textwrap\n    import numpy as np\n\n    seeds = [101,202,303,404,505]\n\n    best_coords = None\n    best_score = -1e9\n\n    for seed in seeds:\n\n        # ⭐ noisy sequence (NEW)\n        noisy_seq = mutate_sequence(sequence, rate=0.01)\n\n        yaml_file = f\"{name}_{seed}.yaml\"\n        out_dir = f\"{name}_{seed}_boltz\"\n\n        if os.path.exists(out_dir):\n            shutil.rmtree(out_dir)\n\n        with open(yaml_file, \"w\") as f:\n            f.write(f\"\"\"\nversion: 1\nsequences:\n  - rna:\n      id: A\n      sequence: {noisy_seq}\n\"\"\")\n\n        wrapper = f\"{name}_{seed}_boltz_wrapper.py\"\n\n        with open(wrapper, \"w\") as f:\n            f.write(textwrap.dedent(f\"\"\"\n                import torch\n                from omegaconf.dictconfig import DictConfig\n\n                torch.serialization.add_safe_globals([DictConfig])\n\n                _orig = torch.load\n                def patched(*a, **k):\n                    k[\"weights_only\"] = False\n                    return _orig(*a, **k)\n                torch.load = patched\n\n                import sys\n                sys.argv = [\n                    \"boltz_runner\",\n                    \"--in_yaml\", \"{yaml_file}\",\n                    \"--out_dir\", \"{out_dir}\",\n                    \"--cache_dir\", \"{BOLTZ_CACHE}\"\n                ]\n\n                import runpy\n                runpy.run_path(\"{BOLTZ_RUNNER}\", run_name=\"__main__\")\n            \"\"\"))\n\n        subprocess.run([\"python\", wrapper])\n\n        cif = find_first_cif(out_dir)\n        coords = extract_c1p(cif)\n\n        if coords is None:\n            continue\n\n        score = boltz_confidence(coords)\n\n        if score > best_score:\n            best_score = score\n            best_coords = coords\n\n    return best_coords","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:01.025847Z","iopub.execute_input":"2026-02-25T14:57:01.026076Z","iopub.status.idle":"2026-02-25T14:57:01.039156Z","shell.execute_reply.started":"2026-02-25T14:57:01.026055Z","shell.execute_reply":"2026-02-25T14:57:01.038436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nos.environ[\"CHAI_DOWNLOADS_DIR\"] = CHAI_WEIGHTS\n\ndef run_chai(sequence, name):\n\n    fasta = f\"{name}.fasta\"\n    out_dir = f\"{name}_chai\"\n\n    with open(fasta, \"w\") as f:\n        f.write(\">rna\\n\")\n        f.write(sequence)\n\n    cmd = [\n        \"python\", CHAI_RUNNER,\n        \"--in_fas\", fasta,\n        \"--out_dir\", out_dir\n    ]\n\n    subprocess.run(cmd)\n\n    return out_dir","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:01.040055Z","iopub.execute_input":"2026-02-25T14:57:01.040382Z","iopub.status.idle":"2026-02-25T14:57:01.055149Z","shell.execute_reply.started":"2026-02-25T14:57:01.040349Z","shell.execute_reply":"2026-02-25T14:57:01.054618Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nfrom pathlib import Path\n\nSRC = \"/kaggle/input/datasets/qiweiyin/protenix-v1-adjusted/Protenix-v1-adjust-v2/Protenix-v1-adjust-v2/Protenix-v1/scripts/msa/data/mmcif\"\nDST = \"/kaggle/working/protenix/mmcif\"\n\nos.makedirs(\"/kaggle/working/protenix\", exist_ok=True)\n\nif not os.path.exists(DST):\n    print(\"Copying mmcif database...\")\n    shutil.copytree(SRC, DST)\nelse:\n    print(\"mmcif already exists\")\n\nprint(\"mmcif ready:\", os.path.exists(DST))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:01.056031Z","iopub.execute_input":"2026-02-25T14:57:01.056286Z","iopub.status.idle":"2026-02-25T14:57:01.125346Z","shell.execute_reply.started":"2026-02-25T14:57:01.056264Z","shell.execute_reply":"2026-02-25T14:57:01.124716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"_PROTENIX_CACHE = {}\n\ndef run_protenix(sequence, name):\n\n    import gc\n    import json\n    import torch\n    import numpy as np\n    import sys\n    from pathlib import Path\n\n    sys.setrecursionlimit(100000)\n\n    seeds = [101, 202, 303]\n    work_dir = Path(\"/kaggle/working\")\n\n    all_predictions = []\n\n    global _PROTENIX_CACHE\n\n    # =========================\n    # BUILD TBM TEMPLATES\n    # =========================\n    templates = find_similar_templates(sequence, top_k=8)\n\n    template_dir = work_dir / f\"{name}_templates\"\n    template_dir.mkdir(exist_ok=True)\n\n    template_paths = []\n\n    ref = None\n\n    for i, t in enumerate(templates):\n\n        mapped = map_template_to_query(sequence, t)\n\n        valid = np.sum(~np.isnan(mapped[:, 0]))\n        if valid < 0.3 * len(sequence):\n            continue\n\n        # ALIGN TO SAME FRAME\n        if ref is None:\n            ref = mapped\n        else:\n            U, Pc, Qc = kabsch(mapped.copy(), ref.copy())\n            if U is not None:\n                mapped = apply_kabsch(mapped, U, Pc, Qc)\n\n        pdb_file = template_dir / f\"template_{i}.pdb\"\n        write_template_pdb(mapped, sequence, pdb_file)\n\n        template_paths.append(str(pdb_file))\n\n    print(\"Templates prepared:\", len(template_paths))\n\n    # =========================\n    # MSA CHECK\n    # =========================\n    MSA_ROOT = \"/kaggle/input/competitions/stanford-rna-3d-folding-2/MSA\"\n    msa_exists = (Path(MSA_ROOT) / f\"{name}.MSA.fasta\").exists()\n\n    if msa_exists:\n        print(\"MSA available for:\", name)\n    else:\n        print(\"⚠️ No MSA found for\", name)\n\n    # =========================\n    # RUN SEEDS\n    # =========================\n    for seed in seeds:\n\n        input_json_path = work_dir / f\"{name}_{seed}_prot.json\"\n\n        data = [{\n            \"name\": name,\n            \"covalent_bonds\": [],\n            \"sequences\": [{\n                \"rnaSequence\": {\n                    \"sequence\": sequence,\n                    \"count\": 1\n                }\n            }]\n        }]\n\n        with open(input_json_path, \"w\") as f:\n            json.dump(data, f)\n\n        dump_dir = work_dir / f\"{name}_{seed}_protenix\"\n\n        cfg = build_protenix_config(\n            str(input_json_path),\n            str(dump_dir),\n            seed=seed\n        )\n\n        cfg = update_gpu_compatible_configs(cfg)\n\n        # ---------- MSA ----------\n        if msa_exists:\n            cfg.use_msa = True\n            cfg.use_rna_msa = True\n            cfg.data.msa.enable_rna_msa = True\n            cfg.data.msa.rna_msadir_raw_paths = [MSA_ROOT]\n        else:\n            cfg.use_msa = False\n            cfg.use_rna_msa = False\n\n        # ---------- TEMPLATE ----------\n        if len(template_paths) > 0:\n            cfg.use_template = True\n            cfg.data.template.enable_template = True\n            cfg.data.template.template_pdb_paths = template_paths\n            cfg.data.template.max_templates = len(template_paths)\n        else:\n            cfg.use_template = False\n\n        try:\n\n            # ---------- Runner cache ----------\n            runner = InferenceRunner(cfg)\n\n            dataset = InferenceDataset(cfg)\n            data_item, atom_array, err = dataset[0]\n\n            if err:\n                continue\n\n            new_cfg = update_inference_configs(\n                cfg,\n                data_item[\"N_token\"].item()\n            )\n\n            # MORE diversity\n            new_cfg.sample_diffusion.N_sample = 96\n\n            runner.update_model_configs(new_cfg)\n\n            pred = runner.predict(data_item)\n\n            raw = pred[\"coordinate\"]\n\n            conf = None\n            if \"confidence\" in pred:\n                conf = pred[\"confidence\"].detach().cpu().numpy().reshape(-1)\n\n            feat = data_item[\"input_feature_dict\"]\n\n            if \"centre_atom_mask\" in feat:\n                mask = feat[\"centre_atom_mask\"] == 1\n            else:\n                mask = feat[\"atom_to_tokatom_idx\"] == 12\n\n            coords_all = raw[:, mask, :].detach().cpu().numpy()\n\n            # =========================\n            # STORE WITH MODEL CONFIDENCE\n            # =========================\n            for idx, c in enumerate(coords_all):\n\n                valid = np.sum(~np.isnan(c[:, 0]))\n                if valid < 3:\n                    continue\n\n                if conf is not None and idx < len(conf):\n                    score = float(conf[idx])\n                else:\n                    score = hybrid_confidence(c)\n\n                all_predictions.append({\n                    \"coords\": c,\n                    \"score\": score\n                })\n\n            del pred, raw, data_item, dataset\n            gc.collect()\n            torch.cuda.empty_cache()\n\n        except Exception as e:\n            print(\"Protenix seed failed:\", seed, e)\n\n    # =========================\n    # NOTHING GENERATED\n    # =========================\n    if len(all_predictions) == 0:\n        return None\n\n    # =========================\n    # SORT BY SCORE (MODEL CONF)\n    # =========================\n    all_predictions = sorted(\n        all_predictions,\n        key=lambda x: x[\"score\"],\n        reverse=True\n    )\n\n    # =========================\n    # DIVERSITY FILTER (IMPORTANT)\n    # =========================\n    def rmsd(a, b):\n        diff = a - b\n        return np.sqrt((diff * diff).sum() / len(a))\n\n    diverse = []\n    RMSD_THRESH = 4.0\n\n    for p in all_predictions:\n\n        keep = True\n\n        for q in diverse:\n            if len(p[\"coords\"]) != len(q[\"coords\"]):\n                continue\n\n            if rmsd(p[\"coords\"], q[\"coords\"]) < RMSD_THRESH:\n                keep = False\n                break\n\n        if keep:\n            diverse.append(p)\n\n        if len(diverse) >= 20:\n            break\n\n    print(\"Protenix kept:\", len(diverse))\n\n    return diverse","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:01.126388Z","iopub.execute_input":"2026-02-25T14:57:01.126706Z","iopub.status.idle":"2026-02-25T14:57:01.142998Z","shell.execute_reply.started":"2026-02-25T14:57:01.126674Z","shell.execute_reply":"2026-02-25T14:57:01.142234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from Bio.PDB.MMCIF2Dict import MMCIF2Dict\nimport numpy as np\n\ndef extract_c1p(cif_file):\n\n    if cif_file is None:\n        return None\n\n    cif = MMCIF2Dict(cif_file)\n\n    atoms = cif[\"_atom_site.label_atom_id\"]\n    xs = cif[\"_atom_site.Cartn_x\"]\n    ys = cif[\"_atom_site.Cartn_y\"]\n    zs = cif[\"_atom_site.Cartn_z\"]\n\n    coords = []\n\n    for a, x, y, z in zip(atoms, xs, ys, zs):\n\n        if a == \"C1'\":\n\n            try:\n                coords.append([float(x), float(y), float(z)])\n            except:\n                continue\n\n    if len(coords) == 0:\n        return None\n\n    return np.array(coords)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:01.146705Z","iopub.execute_input":"2026-02-25T14:57:01.147228Z","iopub.status.idle":"2026-02-25T14:57:01.160874Z","shell.execute_reply.started":"2026-02-25T14:57:01.147204Z","shell.execute_reply":"2026-02-25T14:57:01.160274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# TM SCORE USING USALIGN\n# =========================\n\nimport tempfile\n\ndef write_pdb(coords, fname):\n\n    with open(fname, \"w\") as f:\n        for i,(x,y,z) in enumerate(coords):\n            f.write(\n                f\"ATOM  {i:5d}  C1' A A{i:4d}    \"\n                f\"{x:8.3f}{y:8.3f}{z:8.3f}  1.00  0.00           C\\n\"\n            )\n\n\ndef tm_score_usalign(pred, gt):\n\n    pred = np.asarray(pred)\n    gt = np.asarray(gt)\n\n    # remove invalid residues\n    mask = (\n        ~np.isnan(pred[:, 0]) &\n        ~np.isnan(gt[:, 0])\n    )\n\n    pred = pred[mask]\n    gt = gt[mask]\n\n    if len(pred) < 3 or len(gt) < 3:\n        return None\n\n    with tempfile.TemporaryDirectory() as tmp:\n\n        p1 = os.path.join(tmp, \"a.pdb\")\n        p2 = os.path.join(tmp, \"b.pdb\")\n\n        write_pdb(pred, p1)\n        write_pdb(gt, p2)\n\n        cmd = [\n            USALIGN_BIN,\n            p1,\n            p2,\n            \"-outfmt\", \"2\",\n            \"-atom\", \" C1'\"\n        ]\n\n        res = subprocess.run(cmd, capture_output=True, text=True)\n\n        lines = res.stdout.splitlines()\n\n        if len(lines) < 2:\n            return None\n\n        return float(lines[1].split()[2])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:01.161708Z","iopub.execute_input":"2026-02-25T14:57:01.161988Z","iopub.status.idle":"2026-02-25T14:57:01.17958Z","shell.execute_reply.started":"2026-02-25T14:57:01.161965Z","shell.execute_reply":"2026-02-25T14:57:01.178695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_gt(target_id):\n\n    df = val_labels[val_labels.ID.str.startswith(target_id)]\n\n    coords = df[[\"x_1\",\"y_1\",\"z_1\"]].values\n\n    seq = \"\".join(df.resname.values)\n\n    return coords, seq","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:01.180762Z","iopub.execute_input":"2026-02-25T14:57:01.181131Z","iopub.status.idle":"2026-02-25T14:57:01.194335Z","shell.execute_reply.started":"2026-02-25T14:57:01.181094Z","shell.execute_reply":"2026-02-25T14:57:01.193746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\ndef rmsd(a,b):\n\n    diff = a - b\n    return np.sqrt((diff*diff).sum()/len(a))\n\n\ndef pseudo_tm(pred, gt):\n\n    d = np.linalg.norm(pred - gt, axis=1)\n\n    return np.mean(1/(1 + (d/5)**2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:01.195323Z","iopub.execute_input":"2026-02-25T14:57:01.195558Z","iopub.status.idle":"2026-02-25T14:57:01.209387Z","shell.execute_reply.started":"2026-02-25T14:57:01.195538Z","shell.execute_reply":"2026-02-25T14:57:01.208824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluate(pred_coords, gt_coords):\n\n    if pred_coords is None:\n        return None\n\n    score = tm_score_usalign(pred_coords, gt_coords)\n\n    return score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:01.210169Z","iopub.execute_input":"2026-02-25T14:57:01.210427Z","iopub.status.idle":"2026-02-25T14:57:01.225523Z","shell.execute_reply.started":"2026-02-25T14:57:01.210391Z","shell.execute_reply":"2026-02-25T14:57:01.224993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def patch_with_dl(tbm, dl):\n\n    import numpy as np\n\n    if tbm is None or dl is None:\n        return tbm\n\n    out = tbm.copy()\n\n    L = min(len(out), len(dl))\n\n    for i in range(L):\n        if np.isnan(out[i, 0]):\n            out[i] = dl[i]\n\n    return out","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:01.226235Z","iopub.execute_input":"2026-02-25T14:57:01.226442Z","iopub.status.idle":"2026-02-25T14:57:01.23744Z","shell.execute_reply.started":"2026-02-25T14:57:01.226422Z","shell.execute_reply":"2026-02-25T14:57:01.236831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def ensemble_mean(preds):\n\n    import numpy as np\n\n    if preds is None or len(preds) == 0:\n        return None\n\n    ref = preds[0]\n\n    aligned = []\n\n    for p in preds:\n\n        U, Pc, Qc = kabsch(p.copy(), ref.copy())\n\n        if U is None:\n            continue\n\n        p2 = apply_kabsch(p, U, Pc, Qc)\n        aligned.append(p2)\n\n    if len(aligned) == 0:\n        return ref\n\n    stack = np.stack(aligned)\n\n    return np.nanmedian(stack, axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:01.238876Z","iopub.execute_input":"2026-02-25T14:57:01.239373Z","iopub.status.idle":"2026-02-25T14:57:01.249845Z","shell.execute_reply.started":"2026-02-25T14:57:01.239339Z","shell.execute_reply":"2026-02-25T14:57:01.249254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# TBM TEMPLATE PATH\n# =========================\n\nTBM_ROOT = \"/kaggle/input/datasets/odat1248/rnatbm-templates-20250521-re/fortbm_clustered_sampled_20250526_gt_re\"\n\nimport os\n\nprint(\"TBM root exists:\", os.path.exists(TBM_ROOT))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:01.250654Z","iopub.execute_input":"2026-02-25T14:57:01.250989Z","iopub.status.idle":"2026-02-25T14:57:01.26626Z","shell.execute_reply.started":"2026-02-25T14:57:01.250967Z","shell.execute_reply":"2026-02-25T14:57:01.265532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def map_template_to_query(query_seq, template):\n\n    a, b, _ = better_align(query_seq, template[\"seq\"])\n\n    L = len(query_seq)\n    mapped = np.full((L, 3), np.nan)\n\n    qi = 0\n    ti = 0\n\n    for aa, bb in zip(a, b):\n\n        if aa != \"-\" and bb != \"-\":\n\n            if ti < len(template[\"coords\"]):\n                mapped[qi] = template[\"coords\"][ti]\n\n        if aa != \"-\":\n            qi += 1\n        if bb != \"-\":\n            ti += 1\n\n    return mapped","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:01.267197Z","iopub.execute_input":"2026-02-25T14:57:01.267691Z","iopub.status.idle":"2026-02-25T14:57:01.275948Z","shell.execute_reply.started":"2026-02-25T14:57:01.267657Z","shell.execute_reply":"2026-02-25T14:57:01.27526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# TBM TEMPLATE DATABASE\n# =========================\n\nimport os, glob, json, gzip\nimport numpy as np\n\nTBM_DB = []\n\nfiles = glob.glob(TBM_ROOT + \"/**/*.json*\", recursive=True)\n\nprint(\"Found template json:\", len(files))\n\ndef load_json_coords(json_file):\n\n    try:\n        if json_file.endswith(\".gz\"):\n            data = json.load(gzip.open(json_file, \"rt\"))\n        else:\n            data = json.load(open(json_file))\n\n        key = list(data.keys())[0]\n        residues = data[key]\n\n        seq = []\n        coords = []\n\n        for r in residues:\n\n            seq.append(r[\"one_letter_code\"])\n\n            if \"C1'\" in r[\"atoms\"]:\n                coords.append(r[\"atoms\"][\"C1'\"])\n            else:\n                coords.append([np.nan, np.nan, np.nan])\n\n        return \"\".join(seq), np.array(coords)\n\n    except:\n        return None, None\n\n\nfor f in files:\n\n    seq, coords = load_json_coords(f)\n\n    if seq is None:\n        continue\n\n    TBM_DB.append({\n        \"seq\": seq,\n        \"coords\": coords\n    })\n\nprint(\"Loaded templates:\", len(TBM_DB))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:57:01.276813Z","iopub.execute_input":"2026-02-25T14:57:01.277059Z","iopub.status.idle":"2026-02-25T14:58:50.582965Z","shell.execute_reply.started":"2026-02-25T14:57:01.277038Z","shell.execute_reply":"2026-02-25T14:58:50.582199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def write_template_pdb(coords, seq, out_file):\n\n    with open(out_file, \"w\") as f:\n\n        atom_id = 1\n\n        for i, (res, xyz) in enumerate(zip(seq, coords)):\n\n            if np.isnan(xyz[0]):\n                continue\n\n            x, y, z = xyz\n\n            f.write(\n                f\"ATOM  {atom_id:5d}  C1' {res} A{i:4d}    \"\n                f\"{x:8.3f}{y:8.3f}{z:8.3f}  1.00  0.00           C\\n\"\n            )\n\n            atom_id += 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:58:50.584358Z","iopub.execute_input":"2026-02-25T14:58:50.584957Z","iopub.status.idle":"2026-02-25T14:58:50.589533Z","shell.execute_reply.started":"2026-02-25T14:58:50.584934Z","shell.execute_reply":"2026-02-25T14:58:50.588804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# BETTER ALIGNMENT (BIOPYTHON)\n# =========================\n\nfrom Bio.Align import PairwiseAligner\n\naligner = PairwiseAligner()\naligner.mode = \"global\"\naligner.match_score = 2.0\naligner.mismatch_score = -1.0\naligner.open_gap_score = -3.0\naligner.extend_gap_score = -0.5\n\n\ndef better_align(a, b):\n\n    aln = aligner.align(a, b)[0]\n\n    seqA = aln.target\n    seqB = aln.query\n    score = aln.score\n\n    return seqA, seqB, score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:58:50.590492Z","iopub.execute_input":"2026-02-25T14:58:50.590852Z","iopub.status.idle":"2026-02-25T14:58:50.604103Z","shell.execute_reply.started":"2026-02-25T14:58:50.590813Z","shell.execute_reply":"2026-02-25T14:58:50.603526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# KABSCH ALIGNMENT\n# =========================\n\ndef kabsch(P, Q):\n\n    mask = ~np.isnan(P[:,0]) & ~np.isnan(Q[:,0])\n\n    P = P[mask]\n    Q = Q[mask]\n\n    if len(P) < 3:\n        return None\n\n    Pc = P.mean(axis=0)\n    Qc = Q.mean(axis=0)\n\n    P -= Pc\n    Q -= Qc\n\n    C = P.T @ Q\n    V, S, W = np.linalg.svd(C)\n\n    d = np.sign(np.linalg.det(V @ W))\n\n    D = np.diag([1,1,d])\n\n    U = V @ D @ W\n\n    return U, Pc, Qc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:58:50.604984Z","iopub.execute_input":"2026-02-25T14:58:50.605259Z","iopub.status.idle":"2026-02-25T14:58:50.620236Z","shell.execute_reply.started":"2026-02-25T14:58:50.605227Z","shell.execute_reply":"2026-02-25T14:58:50.619477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def find_similar_templates(query_seq, top_k=8):\n\n    scored = []\n\n    qlen = len(query_seq)\n\n    for t in TBM_DB:\n\n        a, b, score = better_align(query_seq, t[\"seq\"])\n\n        aligned = 0\n        matches = 0\n\n        for x, y in zip(a, b):\n            if x != \"-\" and y != \"-\":\n                aligned += 1\n                if x == y:\n                    matches += 1\n\n        if aligned == 0:\n            continue\n\n        identity = matches / aligned\n        coverage = aligned / max(1, qlen)\n\n        gap_penalty = (len(a) - aligned) / max(1, len(a))\n\n        # HARD FILTERS (very important)\n        if coverage < 0.40:\n            continue\n        if identity < 0.25:\n            continue\n\n        final_score = (\n            identity * 4.0\n            + coverage * 2.0\n            - gap_penalty\n        )\n\n        scored.append((final_score, t))\n\n    scored.sort(key=lambda x: x[0], reverse=True)\n\n    return [t for _, t in scored[:top_k]]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:58:50.621142Z","iopub.execute_input":"2026-02-25T14:58:50.621415Z","iopub.status.idle":"2026-02-25T14:58:50.638676Z","shell.execute_reply.started":"2026-02-25T14:58:50.621385Z","shell.execute_reply":"2026-02-25T14:58:50.63793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def apply_kabsch(coords, U, Pc, Qc):\n\n    return (coords - Pc) @ U + Qc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:58:50.639642Z","iopub.execute_input":"2026-02-25T14:58:50.640036Z","iopub.status.idle":"2026-02-25T14:58:50.652334Z","shell.execute_reply.started":"2026-02-25T14:58:50.640005Z","shell.execute_reply":"2026-02-25T14:58:50.651739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def assemble_templates(query_seq, templates):\n\n    import numpy as np\n\n    L = len(query_seq)\n\n    mapped_list = []\n\n    # ----- helper to map template -----\n    def map_template(t):\n\n        a, b, _ = better_align(query_seq, t[\"seq\"])\n\n        qi = 0\n        ti = 0\n\n        mapped = np.full((L, 3), np.nan)\n\n        for aa, bb in zip(a, b):\n\n            if aa != \"-\" and bb != \"-\":\n                if ti < len(t[\"coords\"]):\n                    mapped[qi] = t[\"coords\"][ti]\n\n            if aa != \"-\":\n                qi += 1\n            if bb != \"-\":\n                ti += 1\n\n        return mapped\n\n\n    # ----- map + filter -----\n    for t in templates:\n\n        mapped = map_template(t)\n\n        valid = np.sum(~np.isnan(mapped[:, 0]))\n        if valid < 5:\n            continue\n\n        # ⭐ confidence filtering (NEW)\n        conf = hybrid_confidence(mapped)\n        if conf < -2.0:\n            continue\n\n        mapped_list.append(mapped)\n\n\n    if len(mapped_list) == 0:\n        return None\n\n\n    # ----- ALIGN ALL TO FIRST TEMPLATE -----\n    ref = mapped_list[0]\n\n    aligned = []\n\n    for m in mapped_list:\n\n        U, Pc, Qc = kabsch(m.copy(), ref.copy())\n\n        if U is None:\n            continue\n\n        m2 = apply_kabsch(m, U, Pc, Qc)\n        aligned.append(m2)\n\n\n    if len(aligned) == 0:\n        return None\n\n\n    # =====================================================\n    # ⭐ PROGRESSIVE MERGE (MAJOR UPGRADE)\n    # =====================================================\n    final = aligned[0].copy()\n\n    for m in aligned[1:]:\n\n        U, Pc, Qc = kabsch(m.copy(), final.copy())\n\n        if U is not None:\n            m = apply_kabsch(m, U, Pc, Qc)\n\n        for i in range(len(final)):\n\n            if np.isnan(final[i, 0]) and not np.isnan(m[i, 0]):\n                final[i] = m[i]\n\n            elif not np.isnan(final[i, 0]) and not np.isnan(m[i, 0]):\n                final[i] = 0.5 * (final[i] + m[i])\n\n    return final","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:58:50.653383Z","iopub.execute_input":"2026-02-25T14:58:50.653672Z","iopub.status.idle":"2026-02-25T14:58:50.667965Z","shell.execute_reply.started":"2026-02-25T14:58:50.653642Z","shell.execute_reply":"2026-02-25T14:58:50.6672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def tbm_predict(sequence):\n\n    if len(TBM_DB) == 0:\n        return None\n\n    # ⭐ fragment modeling hook\n    if len(sequence) > 350:\n        print(\"Long RNA — fragment mode placeholder\")\n        # you can implement fragment version later\n\n    templates = find_similar_templates(sequence, top_k=18)\n\n    candidates = []\n\n    # sliding windows of templates\n    for k in range(8):\n\n        subset = templates[k:k+6]\n\n        if len(subset) == 0:\n            continue\n\n        coords = assemble_templates(sequence, subset)\n\n        if coords is not None:\n            candidates.append(coords)\n\n    if len(candidates) == 0:\n        return None\n\n    return candidates","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:58:50.668931Z","iopub.execute_input":"2026-02-25T14:58:50.669502Z","iopub.status.idle":"2026-02-25T14:58:50.687247Z","shell.execute_reply.started":"2026-02-25T14:58:50.669479Z","shell.execute_reply":"2026-02-25T14:58:50.686467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ndef find_first_cif(folder):\n\n    if folder is None:\n        return None\n\n    for root,_,files in os.walk(folder):\n        for f in files:\n            if f.endswith(\".cif\"):\n                return os.path.join(root,f)\n\n    return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:58:50.688161Z","iopub.execute_input":"2026-02-25T14:58:50.688385Z","iopub.status.idle":"2026-02-25T14:58:50.705377Z","shell.execute_reply.started":"2026-02-25T14:58:50.688364Z","shell.execute_reply":"2026-02-25T14:58:50.704749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nprint(torch.__version__)\nprint(torch.cuda.is_available())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:58:50.70633Z","iopub.execute_input":"2026-02-25T14:58:50.706612Z","iopub.status.idle":"2026-02-25T14:58:50.769041Z","shell.execute_reply.started":"2026-02-25T14:58:50.706583Z","shell.execute_reply":"2026-02-25T14:58:50.768324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\n\n\n\nsys.path.append(PROTENIX_ROOT)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:58:50.770055Z","iopub.execute_input":"2026-02-25T14:58:50.77035Z","iopub.status.idle":"2026-02-25T14:58:50.783754Z","shell.execute_reply.started":"2026-02-25T14:58:50.770327Z","shell.execute_reply":"2026-02-25T14:58:50.783168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def inspect_rdkit():\n\n    print(\"===== RDKit INSPECTION START =====\")\n\n    try:\n        import rdkit\n        from rdkit import Chem\n        from rdkit.Chem import AllChem\n\n        print(\"✅ RDKit import OK\")\n        print(\"RDKit version:\", rdkit.__version__)\n\n    except Exception as e:\n        print(\"❌ RDKit import FAILED:\", e)\n        return\n\n\n    # -------------------------------------------------\n    # Test molecule creation\n    # -------------------------------------------------\n    try:\n        mol = Chem.MolFromSmiles(\"CCO\")\n        assert mol is not None\n        print(\"✅ SMILES parsing OK\")\n    except Exception as e:\n        print(\"❌ SMILES parsing FAILED:\", e)\n\n\n    # -------------------------------------------------\n    # Test 3D embedding\n    # -------------------------------------------------\n    try:\n        mol = Chem.AddHs(mol)\n        AllChem.EmbedMolecule(mol)\n        print(\"✅ 3D embedding OK\")\n    except Exception as e:\n        print(\"❌ 3D embedding FAILED:\", e)\n\n\n    # -------------------------------------------------\n    # Test pickle compatibility (MOST IMPORTANT)\n    # -------------------------------------------------\n    try:\n        import pickle\n\n        data = pickle.dumps(mol)\n        mol2 = pickle.loads(data)\n\n        print(\"✅ Pickle roundtrip OK\")\n\n    except Exception as e:\n        print(\"❌ Pickle FAILED:\", e)\n\n\n    # -------------------------------------------------\n    # Test file IO\n    # -------------------------------------------------\n    try:\n        Chem.MolToMolFile(mol, \"test_rdkit.mol\")\n        mol3 = Chem.MolFromMolFile(\"test_rdkit.mol\")\n\n        if mol3 is not None:\n            print(\"✅ File IO OK\")\n        else:\n            print(\"❌ File IO FAILED\")\n\n    except Exception as e:\n        print(\"❌ File IO FAILED:\", e)\n\n\n    # -------------------------------------------------\n    # Version compatibility hint\n    # -------------------------------------------------\n    try:\n        import rdkit.rdBase as rdBase\n        print(\"RDKit build:\", rdBase.rdkitVersion)\n    except:\n        pass\n\n\n    print(\"===== RDKit INSPECTION END =====\")\n\ninspect_rdkit()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:58:50.784734Z","iopub.execute_input":"2026-02-25T14:58:50.785074Z","iopub.status.idle":"2026-02-25T14:58:50.817308Z","shell.execute_reply.started":"2026-02-25T14:58:50.785051Z","shell.execute_reply":"2026-02-25T14:58:50.816698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom omegaconf.dictconfig import DictConfig\n\n# allow config pickle\ntorch.serialization.add_safe_globals([DictConfig])\n\n# force old behavior\n_orig_load = torch.load\n\ndef patched_load(*args, **kwargs):\n    kwargs[\"weights_only\"] = False\n    return _orig_load(*args, **kwargs)\n\ntorch.load = patched_load\n\nprint(\"Torch patched for Boltz ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:58:50.81817Z","iopub.execute_input":"2026-02-25T14:58:50.818446Z","iopub.status.idle":"2026-02-25T14:58:51.076827Z","shell.execute_reply.started":"2026-02-25T14:58:50.818418Z","shell.execute_reply":"2026-02-25T14:58:51.076058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def hybrid_confidence(coords):\n\n    import numpy as np\n\n    if coords is None:\n        return -1e9\n\n    valid = np.sum(~np.isnan(coords[:, 0]))\n    coverage = valid / len(coords)\n\n    diffs = np.linalg.norm(np.diff(coords, axis=0), axis=1)\n\n    if len(diffs) == 0:\n        return -1e9\n\n    smooth = np.std(diffs)\n    mean_step = np.mean(diffs)\n\n    spacing_error = abs(mean_step - 5.8)\n\n    geom_score = (\n        coverage * 2.0\n        - smooth * 0.3\n        - spacing_error * 0.2\n    )\n\n    # compactness penalty\n    center = coords.mean(axis=0)\n    compact = np.mean(np.linalg.norm(coords - center, axis=1))\n\n    return geom_score - 0.02 * compact","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:58:51.077804Z","iopub.execute_input":"2026-02-25T14:58:51.078177Z","iopub.status.idle":"2026-02-25T14:58:51.083517Z","shell.execute_reply.started":"2026-02-25T14:58:51.078148Z","shell.execute_reply":"2026-02-25T14:58:51.082905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def ensemble_candidates(pred_list):\n\n    import numpy as np\n\n    if pred_list is None or len(pred_list) < 2:\n        return None\n\n    min_len = min(len(p) for p in pred_list)\n    aligned = [p[:min_len] for p in pred_list]\n\n    stack = np.stack(aligned)\n\n    return np.median(stack, axis=0)\n\n\ndef collect_candidates(boltz_coords, prot_preds, tbm_coords):\n\n    candidates = []\n\n    # ---------- TBM FIRST (MOST IMPORTANT) ----------\n    if tbm_coords is not None:\n\n        if isinstance(tbm_coords, list):\n            for t in tbm_coords:\n                if t is not None and len(t) > 0:\n                    candidates.append(t)\n        else:\n            if len(tbm_coords) > 0:\n                candidates.append(tbm_coords)\n\n    # ---------- PROTENIX ----------\n    if prot_preds is not None:\n        for p in prot_preds:\n            if p is None:\n                continue\n            coords = p[\"coords\"]\n            if coords is not None and len(coords) > 0:\n                candidates.append(coords)\n\n    # ---------- BOLTZ ----------\n    if boltz_coords is not None and len(boltz_coords) > 0:\n        candidates.append(boltz_coords)\n\n    return candidates\n\n\ndef competition_best_tm(candidates, gt_coords, top_k=5):\n\n    if candidates is None or len(candidates) == 0:\n        return None\n\n    scores = []\n\n    # assume candidates already ranked\n    for c in candidates[:top_k * 4]:\n\n        if c is None:\n            continue\n\n        s = tm_score_usalign(c, gt_coords)\n\n        if s is None:\n            continue\n\n        scores.append(s)\n\n    if len(scores) == 0:\n        return None\n\n    scores = sorted(scores, reverse=True)[:top_k]\n\n    return max(scores)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:58:51.084658Z","iopub.execute_input":"2026-02-25T14:58:51.084914Z","iopub.status.idle":"2026-02-25T14:58:51.105531Z","shell.execute_reply.started":"2026-02-25T14:58:51.084892Z","shell.execute_reply":"2026-02-25T14:58:51.104946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# COMPETITION EVALUATION LOOP (FINAL — TBM PRIORITY SAFE)\n# =========================\n\nresults = []\n\nMAX_TARGETS = 2\n\niterator = val_seq.iterrows()\n\nif MAX_TARGETS is not None:\n    iterator = val_seq.head(MAX_TARGETS).iterrows()\n\n\nfor i, row in iterator:\n\n    name = row.target_id\n    seq = row.sequence\n\n    print(\"\\n==============================\")\n    print(\"Running target:\", name)\n    print(\"==============================\")\n\n    gt_coords, _ = load_gt(name)\n\n    scores = {\n        \"target\": name,\n        \"competition\": None,\n        \"num_candidates\": 0\n    }\n\n    boltz_coords = None\n    prot_preds = None\n    tbm_coords = None\n\n\n    # =====================================================\n    # PROTENIX\n    # =====================================================\n    try:\n\n        prot_preds = run_protenix(seq, name)\n        print(\"Protenix done\")\n\n        # ---------- PROTENIX CENTROID (SAFE TO KEEP) ----------\n        if prot_preds is not None and len(prot_preds) > 0:\n\n            prot_list = [\n                p[\"coords\"] for p in prot_preds\n                if (p is not None and p.get(\"coords\") is not None)\n            ]\n\n            if len(prot_list) > 1:\n\n                centroid = ensemble_mean(prot_list)\n\n                if centroid is not None:\n\n                    mean_score = np.mean([\n                        p[\"score\"] for p in prot_preds\n                        if p.get(\"score\") is not None\n                    ])\n\n                    prot_preds.append({\n                        \"coords\": centroid,\n                        \"score\": float(mean_score)\n                    })\n\n    except Exception as e:\n        print(\"Protenix failed:\", e)\n\n\n    # =====================================================\n    # BOLTZ\n    # =====================================================\n    try:\n\n        boltz_coords = run_boltz(seq, name)\n        print(\"Boltz done\")\n\n    except Exception as e:\n        print(\"Boltz failed:\", e)\n\n\n    # =====================================================\n    # TBM\n    # =====================================================\n    try:\n\n        tbm_coords = tbm_predict(seq)\n        print(\"TBM done\")\n\n        # ---------- TBM CENTROID DISABLED ----------\n        # (kept raw TBM only to preserve accuracy)\n\n    except Exception as e:\n        print(\"TBM failed:\", e)\n\n\n    # =====================================================\n    # PATCH TBM WITH DL — DISABLED (CRITICAL FIX)\n    # =====================================================\n    if False:\n        if tbm_coords is not None and prot_preds is not None:\n\n            try:\n\n                best_pred = max(\n                    prot_preds,\n                    key=lambda x: (\n                        x.get(\"score\") if (x and x.get(\"score\") is not None)\n                        else -1e9\n                    )\n                )\n\n                best_dl = best_pred[\"coords\"]\n\n                patched = []\n\n                tbm_list = (\n                    tbm_coords if isinstance(tbm_coords, list)\n                    else [tbm_coords]\n                )\n\n                for t in tbm_list[:5]:\n\n                    if t is None:\n                        continue\n\n                    patched_coords = patch_with_dl(t, best_dl)\n\n                    if patched_coords is not None:\n                        patched.append(patched_coords)\n\n                if len(patched) > 0:\n\n                    if isinstance(tbm_coords, list):\n                        tbm_coords.extend(patched)\n                    else:\n                        tbm_coords = [tbm_coords] + patched\n\n            except Exception as e:\n                print(\"TBM patch failed:\", e)\n\n\n    # =====================================================\n    # COLLECT CANDIDATES  (YOUR FIX 2 SHOULD BE HERE)\n    # =====================================================\n    candidates = collect_candidates(\n        boltz_coords,\n        prot_preds,\n        tbm_coords\n    )\n\n\n    # =====================================================\n    # CLEAN INVALID\n    # =====================================================\n    clean_candidates = []\n\n    for c in candidates:\n\n        if c is None:\n            continue\n\n        valid = np.sum(~np.isnan(c[:, 0]))\n\n        if valid < 5:\n            continue\n\n        clean_candidates.append(c)\n\n    candidates = clean_candidates\n\n    scores[\"num_candidates\"] = len(candidates)\n\n    print(\"Candidates:\", len(candidates))\n\n\n    # =====================================================\n    # RANK BY CONFIDENCE — REMOVED (CRITICAL FIX)\n    # =====================================================\n    # KEEP ORIGINAL ORDER (TBM FIRST)\n    pass\n\n\n    # =====================================================\n    # GLOBAL RMSD DIVERSITY FILTER — DISABLED\n    # =====================================================\n    diverse = candidates\n    candidates = diverse\n\n\n    # =====================================================\n    # SCORE\n    # =====================================================\n    try:\n\n        best_tm = competition_best_tm(\n            candidates,\n            gt_coords,\n            top_k=5\n        )\n\n        scores[\"competition\"] = best_tm\n\n    except Exception as e:\n        print(\"Scoring failed:\", e)\n        scores[\"competition\"] = None\n\n\n    print(\"Scores:\", scores)\n\n    results.append(scores)\n\n\n# =====================================================\n# SUMMARY\n# =====================================================\nimport pandas as pd\n\ndf = pd.DataFrame(results)\n\nprint(\"\\nFINAL MEAN\")\nprint(df.mean(numeric_only=True))\n\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T14:58:51.106541Z","iopub.execute_input":"2026-02-25T14:58:51.106862Z","iopub.status.idle":"2026-02-25T15:47:01.527421Z","shell.execute_reply.started":"2026-02-25T14:58:51.10683Z","shell.execute_reply":"2026-02-25T15:47:01.52684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndf = pd.DataFrame(results)\n\nprint(df.select_dtypes(include=\"number\").mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T15:47:01.528404Z","iopub.execute_input":"2026-02-25T15:47:01.528647Z","iopub.status.idle":"2026-02-25T15:47:01.534509Z","shell.execute_reply.started":"2026-02-25T15:47:01.528626Z","shell.execute_reply":"2026-02-25T15:47:01.533972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}