{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":118765,"databundleVersionId":15231210,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":11775065,"sourceType":"datasetVersion","datasetId":7392749},{"sourceId":14429909,"sourceType":"datasetVersion","datasetId":9216133}],"dockerImageVersionId":31234,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":6.351114,"end_time":"2026-01-07T22:42:08.555893","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-01-07T22:42:02.204779","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport subprocess\nimport sys\nimport logging\n\n# 配置日誌\nlogging.basicConfig(level=logging.INFO, format='%(asctime)s | %(message)s')\nlogger = logging.getLogger(\"RNA_Pipeline\")\n\ndef deploy_offline_packages(package_path):\n    \"\"\"\n    執行 Python 3.12 的離線安裝流程\n    \"\"\"\n    if not os.path.exists(package_path):\n        logger.error(f\"找不到路徑: {package_path}\")\n        return False\n    \n    logger.info(\"⚙️ 正在安裝 Biopython 與 ViennaRNA (離線模式)...\")\n    try:\n        subprocess.check_call([\n            sys.executable, \"-m\", \"pip\", \"install\", \n            \"--no-index\", \"--find-links\", package_path, \n            \"biopython\", \"viennarna\"\n        ])\n        logger.info(\"✅ 安裝成功！\")\n        return True\n    except Exception as e:\n        logger.error(f\"❌ 安裝失敗: {e}\")\n        return False\n\n# 定義路徑\nPACKAGE_DIR = \"/kaggle/input/rna-offline-packages/rna-packages-312\"\ndeploy_offline_packages(PACKAGE_DIR)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\nStanford RNA 3D Folding Part 2 - Gold Standard v2.1\nStrategy: 0.177 Baseline Logic + 93-Mapping + Rigid Rod Fallback (No Damping)\nAuthor: AI Architecture Team (AIPE02 Stability Task)\n\"\"\"\n\nimport os\nimport sys\nimport re\nimport shutil\nimport subprocess\nimport warnings\nimport logging\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom multiprocessing import Pool\nfrom Bio import Align\nfrom Bio.PDB import MMCIFParser\nfrom Bio.PDB.PDBExceptions import PDBConstructionWarning\n\n# ==========================================\n# 0. 環境與日誌\n# ==========================================\nlogging.basicConfig(level=logging.INFO, format='%(asctime)s | %(levelname)s | %(message)s', datefmt='%H:%M:%S')\nlogger = logging.getLogger(\"GoldV2.1\")\nwarnings.simplefilter('ignore', PDBConstructionWarning)\n\n# ==========================================\n# 1. 系統配置 (對標 0.177 高穩定參數)\n# ==========================================\nclass Config:\n    BASE_DIR = Path('/kaggle/input/stanford-rna-3d-folding-2')\n    MMSEQS_INPUT = Path('/kaggle/input/mmseqs2/mmseqs/bin/mmseqs')\n    WORK_DIR = Path('/kaggle/working')\n    MMSEQS_BIN = WORK_DIR / 'mmseqs'\n    \n    CIF_DIR = BASE_DIR / 'PDB_RNA'\n    METADATA_PATH = BASE_DIR / 'extra/rna_metadata.csv'\n    TEST_SEQ_PATH = BASE_DIR / 'test_sequences.csv'\n    SUBMISSION_PATH = 'submission.csv'\n    \n    NUM_MODELS = 5\n    N_CORES = 4\n    \n    # 物理參數\n    IDEAL_DIST = 5.9\n    SENSITIVITY = 7.5 # 對標第一名的高敏感度搜尋 (官方預設通常較低)\n    REPAIR_THRESHOLD = 15.0 # 僅修復嚴重斷裂\n\n# ==========================================\n# 2. 數據映射 (93-Variant Implementation)\n# ==========================================\nclass RNAUtils:\n    # 確保座標提取的最大化，涵蓋 PNAS 提到的常見修飾\n    MAPPING = {\n        'A': 'A', 'U': 'U', 'G': 'G', 'C': 'C', \n        'I': 'A', '1MA': 'A', '2MA': 'A', '6MA': 'A', 'M2A': 'A', 'MS2': 'A', 'AET': 'A',\n        'PSU': 'U', 'H2U': 'U', '5MU': 'U', '4SU': 'U', 'OMU': 'U', 'T': 'U', 'DHU': 'U',\n        'M2G': 'G', 'M7G': 'G', 'OMG': 'G', '1MG': 'G', '2MG': 'G', 'QUO': 'G', 'G7M': 'G',\n        '5MC': 'C', 'OMC': 'C', 'M5C': 'C', 'CBV': 'C', 'S2C': 'C',\n        'D': 'U', 'P': 'U', 'Q': 'G', 'RT': 'U'\n    }\n\n    @staticmethod\n    def canonicalize(res):\n        return RNAUtils.MAPPING.get(res.strip().upper(), 'X')\n\n    @staticmethod\n    def get_chain_boundaries(row):\n        \"\"\"\n        解析鏈邊界，防止將不同鏈強行連接\n        雖然 0.177 版本不處理這個，但在 v2.1 加入此保護能避免物理錯誤扣分\n        \"\"\"\n        try:\n            if pd.isna(row['stoichiometry']) or pd.isna(row['all_sequences']): return set()\n            \n            chain_lengths = {}\n            lines = row['all_sequences'].strip().split('\\n')\n            current_len = 0\n            current_ids = []\n            \n            for line in lines:\n                if line.startswith('>'):\n                    parts = line.split(\"|\")\n                    if len(parts) >= 2:\n                        ids_part = re.sub(r\"^Chains? \", \"\", parts[1].strip())\n                        current_ids = [re.search(r\"\\[auth ([^\\]]+)\\]\", c).group(1) if \"[\" in c else c.strip() for c in ids_part.split(\",\")]\n                    current_len = 0\n                else:\n                    current_len += len(line.strip())\n                    for cid in current_ids: chain_lengths[cid] = current_len\n\n            stoich_str = row['stoichiometry'].strip(\"{}\")\n            curr_idx = 0\n            boundaries = set()\n            for item in stoich_str.split(\";\"):\n                if \":\" not in item: continue\n                cid, count = item.split(\":\")\n                cid = cid.strip()\n                if cid in chain_lengths:\n                    slen = chain_lengths[cid]\n                    for _ in range(int(count)):\n                        curr_idx += slen\n                        boundaries.add(curr_idx - 1)\n            \n            if (curr_idx - 1) in boundaries: boundaries.remove(curr_idx - 1)\n            return boundaries\n        except: return set()\n\n# ==========================================\n# 3. 核心推理引擎 (剛性幾何 GoldEngine)\n# ==========================================\nclass GoldEngine:\n    def __init__(self, metadata, hits):\n        self.metadata = metadata\n        self.hits = hits\n        self.parser = MMCIFParser(QUIET=True)\n        self.aligner = Align.PairwiseAligner()\n        self.aligner.mode = 'global'\n        self.aligner.match_score = 3      # 強化匹配權重\n        self.aligner.open_gap_score = -0.5 # 允許較小的 Gap\n\n    def _generate_rigid_rod(self, L, m_idx):\n        \"\"\"\n        產生剛性、無扭曲的直線螺旋 (A-form Helix)\n        這是比 0.0 更強的 Baseline\n        \"\"\"\n        coords = np.zeros((L, 3))\n        rise, radius = 2.8, 9.0\n        # 僅進行相位偏移 (Phase Shift)，不進行任何向心收縮 (No Damping)\n        phase = m_idx * 0.5 \n        \n        for i in range(L):\n            angle = i * np.deg2rad(32.7) + phase\n            # 純粹的螺旋方程式\n            coords[i] = [\n                radius * np.cos(angle), \n                radius * np.sin(angle), \n                i * rise\n            ]\n        return coords\n\n    def _get_coords(self, pdb, chain):\n        \"\"\"\n        從 CIF 檔案讀取座標，應用 93-Mapping\n        \"\"\"\n        path = Config.CIF_DIR / f\"{pdb.lower().split('_')[0]}.cif\"\n        if not path.exists(): return None, \"\"\n        try:\n            struct = self.parser.get_structure('R', str(path))\n            coords, seq_list = {}, []\n            for model in struct:\n                if chain in model:\n                    for res in model[chain]:\n                        if \"C1'\" in res:\n                            # 關鍵優化：使用廣義映射表\n                            clean = RNAUtils.canonicalize(res.get_resname())\n                            coords[res.id[1]] = res[\"C1'\"].get_coord()\n                            seq_list.append((res.id[1], clean))\n                    break\n            # 確保殘基順序正確\n            sorted_seq = sorted(seq_list, key=lambda x: x[0])\n            return coords, \"\".join([x[1] for x in sorted_seq])\n        except: return None, \"\"\n\n    def process(self, task):\n        idx, row = task\n        tid, seq = row['target_id'], row['sequence']\n        L = len(seq)\n        boundaries = RNAUtils.get_chain_boundaries(row)\n        \n        hits = self.hits.get(tid, [])\n        models = []\n\n        for m in range(Config.NUM_MODELS):\n            xyz, filled = np.zeros((L, 3)), np.zeros(L, dtype=bool)\n            \n            # 策略 A: 範本映射 (Template Mapping)\n            if m < len(hits):\n                hit = hits[m]\n                pdb_c, obs_seq = self._get_coords(hit['pdb_id'], hit['chain_id'])\n                if pdb_c and obs_seq:\n                    try:\n                        # 執行全域對齊\n                        aln = self.aligner.align(seq, obs_seq)[0]\n                        t_idx, s_idx = aln.aligned\n                        resids = sorted(pdb_c.keys())\n                        \n                        # 填入座標\n                        for tr, sr in zip(t_idx, s_idx):\n                            for t, s in zip(range(*tr), range(*sr)):\n                                if s < len(resids):\n                                    rid = resids[s]\n                                    if rid in pdb_c:\n                                        xyz[t] = pdb_c[rid]\n                                        filled[t] = True\n                    except: pass\n\n            # 策略 B: 剛性螺旋補全 (Rigid Rod Fallback)\n            # 取代 0.177 版的 0.0 填充，給出物理分母\n            if not np.all(filled):\n                rod = self._generate_rigid_rod(L, m)\n                xyz[~filled] = rod[~filled]\n            \n            # 策略 C: 極簡物理修復 (Minimal Repair)\n            # 僅針對明顯斷裂 (Gap) 進行平移，不做額外的能量最小化，避免破壞範本幾何\n            for i in range(L-1):\n                if i in boundaries: continue # 跳過鏈邊界\n                \n                d = np.linalg.norm(xyz[i+1] - xyz[i])\n                # 只有當距離大於 15A (嚴重斷裂) 時才修復，小幅偏差予以保留\n                if d > Config.REPAIR_THRESHOLD:\n                    direction = (xyz[i+1] - xyz[i]) / (d + 1e-10)\n                    # 將斷裂後的所有原子拉回，維持 5.9A 間距\n                    shift = (xyz[i] + direction * Config.IDEAL_DIST) - xyz[i+1]\n                    xyz[i+1:] += shift\n            \n            models.append(xyz)\n            \n        return tid, seq, models\n\n# ==========================================\n# 4. 主流程\n# ==========================================\ndef main():\n    logger.info(\"💎 Gold Standard v2.1 Pipeline Started...\")\n    \n    # 1. MMseqs2 佈署 (確保二進位檔可用)\n    if Config.MMSEQS_INPUT.exists() and not Config.MMSEQS_BIN.exists():\n        os.makedirs(Config.MMSEQS_BIN.parent, exist_ok=True)\n        shutil.copy(Config.MMSEQS_INPUT, Config.MMSEQS_BIN)\n        os.chmod(Config.MMSEQS_BIN, 0o755)\n\n    test_df = pd.read_csv(Config.TEST_SEQ_PATH)\n    meta_df = pd.read_csv(Config.METADATA_PATH)\n\n    # 2. 執行高敏感度搜尋\n    hits = {}\n    if Config.MMSEQS_BIN.exists():\n        qf, tf, rf = 'q.fasta', 't.fasta', 'res.m8'\n        logger.info(\"🔍 Running MMseqs2 Search (Sensitivity: 7.5)...\")\n        \n        with open(qf, 'w') as f:\n            for _, r in test_df.iterrows(): f.write(f\">{r['target_id']}\\n{r['sequence']}\\n\")\n        \n        # 僅使用結構化程度 > 0.4 的高品質 RNA 作為資料庫\n        valid_meta = meta_df[meta_df['total_structuredness_adjusted'] > 0.4]\n        with open(tf, 'w') as f:\n            for _, r in valid_meta.iterrows(): f.write(f\">{r['pdb_id']}|{r['auth_chain_id']}\\n{r['sequence']}\\n\")\n        \n        # 呼叫 MMseqs2\n        subprocess.run([str(Config.MMSEQS_BIN), 'easy-search', qf, tf, rf, 'tmp', \n                        '--format-output', 'query,target,fident', \n                        '-s', str(Config.SENSITIVITY), \n                        '--threads', str(Config.N_CORES)], \n                       stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)\n        \n        # 解析結果\n        if Path(rf).exists():\n            df_r = pd.read_csv(rf, sep='\\t', names=['q', 't', 'f'])\n            for q, g in df_r.groupby('q'):\n                hits[q] = [{'pdb_id': r['t'].split('|')[0], 'chain_id': r['t'].split('|')[1], 'score': r['f']} \n                          for _, r in g.sort_values('f', ascending=False).head(10).iterrows()]\n\n    # 3. 啟動推理引擎\n    engine = GoldEngine(meta_df, hits)\n    tasks = [(i, r) for i, r in test_df.iterrows()]\n    \n    logger.info(f\"⚙️ Processing {len(tasks)} targets...\")\n    with Pool(Config.N_CORES) as pool:\n        results = pool.map(engine.process, tasks)\n\n    # 4. 格式化輸出\n    final_rows = []\n    for tid, seq, models in results:\n        for i, char in enumerate(seq):\n            row = {'ID': f\"{tid}_{i+1}\", 'resname': char, 'resid': i+1}\n            for m in range(5):\n                # 數值裁剪與四捨五入\n                xyz = np.clip(models[m][i], -999.999, 9999.999)\n                row[f'x_{m+1}'], row[f'y_{m+1}'], row[f'z_{m+1}'] = np.round(xyz, 3)\n            final_rows.append(row)\n    \n    pd.DataFrame(final_rows).to_csv(Config.SUBMISSION_PATH, index=False)\n    logger.info(f\"✅ Pipeline Completed Successfully. File: {Config.SUBMISSION_PATH}\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T02:38:05.301692Z","iopub.execute_input":"2026-01-08T02:38:05.302024Z","execution_failed":"2026-01-08T02:38:47.861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# import re\n# import numpy as np\n# import pandas as pd\n# from scipy.spatial.transform import Rotation\n\n# class RNAMultimerPredictor:\n#     \"\"\"\n#     針對 RNA 多鏈複合物設計的結構預測器\n#     支援：多鏈解析、殘基編號對齊、多狀態隨機擾動\n#     \"\"\"\n#     def __init__(self):\n#         # A-form 螺旋標準參數\n#         self.params = {\n#             'rise': 2.8,\n#             'twist': np.deg2rad(32.7),\n#             'radius': 9.0\n#         }\n\n#     def parse_stoichiometry(self, stoich_str):\n#         \"\"\"\n#         解析 stoichiometry 欄位，例如 \"{A:1;B:2}\" -> [('A', 1), ('B', 2)]\n#         \"\"\"\n#         # 移除大括號並分割\n#         content = stoich_str.strip(\"{}\")\n#         items = content.split(\";\")\n#         parsed = []\n#         for item in items:\n#             if \":\" in item:\n#                 chain_id, count = item.split(\":\")\n#                 parsed.append((chain_id, int(count)))\n#         return parsed\n\n#     def generate_helix(self, length, seed_offset=0):\n#         \"\"\"\n#         生成單條螺旋鏈的 C1' 座標\n#         \"\"\"\n#         coords = []\n#         # 加入隨機擾動以生成不同狀態 (States)\n#         current_rise = self.params['rise'] * (0.95 + 0.1 * np.random.rand())\n#         current_twist = self.params['twist'] * (0.95 + 0.1 * np.random.rand())\n        \n#         # 隨機初始旋轉\n#         initial_rotation = Rotation.random().as_matrix()\n        \n#         for i in range(length):\n#             angle = i * current_twist + seed_offset\n#             x = self.params['radius'] * np.cos(angle)\n#             y = self.params['radius'] * np.sin(angle)\n#             z = i * current_rise\n#             coords.append([x, y, z])\n            \n#         return np.array(coords) @ initial_rotation.T\n\n#     def predict(self, row, num_models=5):\n#         \"\"\"\n#         為單一 target 生成 5 組預測結果\n#         \"\"\"\n#         target_id = row['target_id']\n#         full_sequence = row['sequence']\n#         stoich_list = self.parse_stoichiometry(row['stoichiometry'])\n        \n#         # 1. 拆分序列：假設 sequence 是按照 stoichiometry 順序拼接的\n#         # 注意：實際比賽可能需要更複雜的 all_sequences 解析，此處為簡化邏輯\n#         models_coords = []\n        \n#         for m in range(num_models):\n#             all_chain_coords = []\n#             current_pos = 0\n            \n#             # 為每一種鏈分別生成\n#             for chain_id, count in stoich_list:\n#                 # 這裡假設每種 chain 的長度均等（簡易切分）\n#                 # 建議實務上配合 all_sequences 獲取準確長度\n#                 avg_len = len(full_sequence) // sum(c for _, c in stoich_list)\n                \n#                 for c_idx in range(count):\n#                     helix = self.generate_helix(avg_len, seed_offset=m + c_idx)\n#                     # 空間位移，避免鏈與鏈重疊\n#                     helix += np.array([c_idx * 25.0, m * 20.0, 0]) \n#                     all_chain_coords.append(helix)\n            \n#             # 合併所有鏈的座標\n#             combined = np.vstack(all_chain_coords)\n#             # 若因切分導致長度不符，進行填充或截斷\n#             if len(combined) < len(full_sequence):\n#                 padding = np.zeros((len(full_sequence) - len(combined), 3))\n#                 combined = np.vstack([combined, padding])\n#             else:\n#                 combined = combined[:len(full_sequence)]\n                \n#             models_coords.append(combined)\n            \n#         return models_coords\n\n# def run_pipeline(input_path, output_path):\n#     df = pd.read_csv(input_path)\n#     predictor = RNAMultimerPredictor()\n#     submission_rows = []\n\n#     print(f\"Processing {len(df)} sequences...\")\n\n#     for _, row in df.iterrows():\n#         target_id = row['target_id']\n#         sequence = row['sequence']\n        \n#         # 生成 5 個模型\n#         models = predictor.predict(row, num_models=5)\n        \n#         for i, char in enumerate(sequence):\n#             # 建立符合格式的 Row\n#             # ID 格式: target_id_resid\n#             res_data = {\n#                 'ID': f\"{target_id}_{i+1}\",\n#                 'resname': char,\n#                 'resid': i + 1\n#             }\n            \n#             for m_idx in range(5):\n#                 coords = models[m_idx][i]\n#                 res_data[f'x_{m_idx+1}'] = np.clip(coords[0], -999.999, 9999.999)\n#                 res_data[f'y_{m_idx+1}'] = np.clip(coords[1], -999.999, 9999.999)\n#                 res_data[f'z_{m_idx+1}'] = np.clip(coords[2], -999.999, 9999.999)\n            \n#             submission_rows.append(res_data)\n\n#     # 建立 DataFrame 並排序欄位\n#     sub_df = pd.DataFrame(submission_rows)\n#     cols = ['ID', 'resname', 'resid']\n#     for i in range(1, 6):\n#         cols += [f'x_{i}', f'y_{i}', f'z_{i}']\n    \n#     sub_df = sub_df[cols]\n#     sub_df.to_csv(output_path, index=False, float_format='%.3f')\n#     print(f\"Submission saved to {output_path}\")\n\n# # 執行\n# if __name__ == \"__main__\":\n#     INPUT_CSV = '/kaggle/input/stanford-rna-3d-folding-2/test_sequences.csv'\n#     if os.path.exists(INPUT_CSV):\n#         run_pipeline(INPUT_CSV, 'submission.csv')\n#     else:\n#         # 測試模式\n#         print(\"Input file not found, running with dummy data...\")","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2026-01-08T02:29:38.196238Z","iopub.execute_input":"2026-01-08T02:29:38.196587Z","iopub.status.idle":"2026-01-08T02:29:38.204311Z","shell.execute_reply.started":"2026-01-08T02:29:38.196559Z","shell.execute_reply":"2026-01-08T02:29:38.203333Z"},"papermill":{"duration":2.559739,"end_time":"2026-01-07T22:42:08.032841","exception":false,"start_time":"2026-01-07T22:42:05.473102","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import pandas as pd\n# from difflib import SequenceMatcher\n\n# class RNATemplateSearcher:\n#     \"\"\"\n#     基於序列相似度的 3D 範本檢索系統\n#     \"\"\"\n#     def __init__(self, metadata_path):\n#         # 僅載入必要欄位以節省記憶體\n#         self.df = pd.read_csv(metadata_path, usecols=['target_id', 'sequence', 'total_structuredness_adjusted'])\n#         # 過濾掉結構化程度過低的範本\n#         self.df = self.df[self.df['total_structuredness_adjusted'] > 0.6]\n\n#     def get_similarity(self, seq1, seq2):\n#         \"\"\"\n#         計算兩條序列的相似度 (使用快速比對演算法)\n#         \"\"\"\n#         return SequenceMatcher(None, seq1, seq2).ratio()\n\n#     def search(self, target_sequence, top_n=5, min_threshold=0.3):\n#         \"\"\"\n#         檢索最相似的範本\n#         \"\"\"\n#         results = []\n#         target_len = len(target_sequence)\n        \n#         # 為了效能，先過濾長度相近的範本 (+/- 30%)\n#         potential_matches = self.df[\n#             (self.df['sequence'].str.len() > target_len * 0.7) & \n#             (self.df['sequence'].str.len() < target_len * 1.3)\n#         ]\n        \n#         for _, row in potential_matches.iterrows():\n#             sim = self.get_similarity(target_sequence, row['sequence'])\n#             if sim >= min_threshold:\n#                 results.append({\n#                     'template_id': row['target_id'],\n#                     'similarity': sim,\n#                     'sequence': row['sequence']\n#                 })\n        \n#         # 按相似度排序\n#         results.sort(key=lambda x: x['similarity'], reverse=True)\n#         return results[:top_n]\n\n# # 使用範例\n# searcher = RNATemplateSearcher('/kaggle/input/stanford-rna-3d-folding-2/extra/rna_metadata.csv')\n# target_seq = \"ACCGUGACGGGCCUUUUGGCUAUACGCGGU\" # 測試集 8ZNQ 的序列\n# best_templates = searcher.search(target_seq)\n# for t in best_templates:\n#     print(f\"找到範本: {t['template_id']} | 相似度: {t['similarity']:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T02:29:38.206709Z","iopub.execute_input":"2026-01-08T02:29:38.207525Z","iopub.status.idle":"2026-01-08T02:29:38.224677Z","shell.execute_reply.started":"2026-01-08T02:29:38.20748Z","shell.execute_reply":"2026-01-08T02:29:38.223638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import numpy as np\n# from Bio import SeqIO\n# from pathlib import Path\n\n# def analyze_msa_coevolution(msa_path, top_k=20):\n#     \"\"\"\n#     分析 MSA 檔案並回傳最強的共進化殘基對 (可能是鹼基對)\n#     \"\"\"\n#     records = list(SeqIO.parse(msa_path, \"fasta\"))\n#     if not records: return []\n    \n#     # 轉換為矩陣 (N_seq x L_length)\n#     msa_mat = np.array([list(str(r.seq)) for r in records])\n#     N, L = msa_mat.shape\n    \n#     # 簡化計算：只計算 A, C, G, U\n#     # 這裡建議使用 AIPE 課程中學過的矩陣向量化來加速\n#     alphabet = ['A', 'C', 'G', 'U']\n#     char_map = {c: i for i, c in enumerate(alphabet)}\n    \n#     # 計算 MI 矩陣 (簡化版邏輯)\n#     mi_matrix = np.zeros((L, L))\n#     for i in range(L):\n#         for j in range(i + 4, L): # 過濾相鄰殘基\n#             # 計算機率分佈 (這裡實際執行時建議使用 Numpy bincount)\n#             # ... (此處省略部分高效計算邏輯以節省空間)\n#             pass\n            \n#     print(f\"Target: {Path(msa_path).stem} | Sequences: {N} | Length: {L}\")\n#     return \"MSA 分析完成，準備回傳共進化特徵...\"\n\n# analyze_msa_coevolution('/kaggle/input/stanford-rna-3d-folding-2/MSA/157D.MSA.fasta')\n# analyze_msa_coevolution('/kaggle/input/stanford-rna-3d-folding-2/MSA/1A34.MSA.fasta')\n# analyze_msa_coevolution('/kaggle/input/stanford-rna-3d-folding-2/MSA/1AQ4.MSA.fasta')","metadata":{"papermill":{"duration":0.001438,"end_time":"2026-01-07T22:42:08.035969","exception":false,"start_time":"2026-01-07T22:42:08.034531","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T02:29:38.226082Z","iopub.execute_input":"2026-01-08T02:29:38.226501Z","iopub.status.idle":"2026-01-08T02:29:38.245178Z","shell.execute_reply.started":"2026-01-08T02:29:38.226461Z","shell.execute_reply":"2026-01-08T02:29:38.244144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from Bio.PDB import MMCIFParser\n# import pandas as pd\n\n# def robust_extract_c1_prime(cif_path):\n#     parser = MMCIFParser(QUIET=True)\n#     try:\n#         structure = parser.get_structure('RNA', cif_path)\n#     except Exception as e:\n#         return f\"讀取失敗: {e}\"\n    \n#     data = []\n#     # 定義常見的 RNA 殘基名稱 (包含修飾過的)\n#     rna_residues = ['A', 'C', 'G', 'U', 'RA', 'RC', 'RG', 'RU']\n    \n#     for model in structure:\n#         for chain in model:\n#             for residue in chain:\n#                 resname = residue.get_resname().strip().upper()\n#                 # 只要包含 C1' 就嘗試抓取，不論殘基名稱\n#                 if \"C1'\" in residue:\n#                     atom = residue[\"C1'\"]\n#                     coord = atom.get_coord()\n#                     data.append({\n#                         'chain': chain.id,\n#                         'resname': resname,\n#                         'resid': residue.id[1],\n#                         'x': coord[0], 'y': coord[1], 'z': coord[2]\n#                     })\n    \n#     df = pd.DataFrame(data)\n#     if df.empty:\n#         print(f\"⚠️ 警告: 在 {cif_path} 中找不到任何 C1' 原子。\")\n#     else:\n#         print(f\"✅ 成功提取 {len(df)} 個 C1' 座標。\")\n#     return df\n\n# # 測試 100d.cif\n# df_coords = robust_extract_c1_prime('/kaggle/input/stanford-rna-3d-folding-2/PDB_RNA/100d.cif')\n# print(df_coords.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T02:29:38.246602Z","iopub.execute_input":"2026-01-08T02:29:38.247489Z","iopub.status.idle":"2026-01-08T02:29:38.263729Z","shell.execute_reply.started":"2026-01-08T02:29:38.247459Z","shell.execute_reply":"2026-01-08T02:29:38.262855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import pandas as pd\n\n# def analyze_rna_metadata(file_path):\n#     # 使用 chunk 處理大型 CSV 以節省記憶體 (AIPE 課程中的大數據處理技巧)\n#     cols_to_use = ['target_id', 'total_structuredness_adjusted', 'length', 'fraction_observed']\n    \n#     print(\"--- 正在分析 Metadata ---\")\n#     df = pd.read_csv(file_path, usecols=cols_to_use)\n    \n#     # 1. 計算平均結構化程度\n#     mean_struct = df['total_structuredness_adjusted'].mean()\n    \n#     # 2. 分析長度分佈\n#     length_bins = pd.cut(df['length'], bins=[0, 100, 500, 1000, 5500])\n#     dist = df.groupby(length_bins, observed=True).size()\n    \n#     print(f\"🔹 平均結構化程度 (Total Structuredness Adjusted): {mean_struct:.4f}\")\n#     print(f\"🔹 數據集中各長度區間的數量:\\n{dist}\")\n    \n#     # 3. 找出高品質範本 (結構化程度 > 0.8 且觀測比例高)\n#     high_quality = df[(df['total_structuredness_adjusted'] > 0.8) & (df['fraction_observed'] > 0.9)]\n#     print(f\"🔹 高品質範本數量: {len(high_quality)}\")\n    \n#     return mean_struct\n\n# # 執行分析\n# METADATA_PATH = '/kaggle/input/stanford-rna-3d-folding-2/extra/rna_metadata.csv'\n# analyze_rna_metadata(METADATA_PATH)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T02:29:38.264993Z","iopub.execute_input":"2026-01-08T02:29:38.26539Z","iopub.status.idle":"2026-01-08T02:29:38.282423Z","shell.execute_reply.started":"2026-01-08T02:29:38.265351Z","shell.execute_reply":"2026-01-08T02:29:38.281396Z"}},"outputs":[],"execution_count":null}]}