{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":99552,"databundleVersionId":13747926,"sourceType":"competition"},{"sourceId":13073209,"sourceType":"datasetVersion","datasetId":7979194}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !python -m pip install -U \"pylibjpeg[libjpeg]\" pylibjpeg-libjpeg","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:36:50.682234Z","iopub.execute_input":"2025-09-16T06:36:50.682586Z","iopub.status.idle":"2025-09-16T06:36:50.688852Z","shell.execute_reply.started":"2025-09-16T06:36:50.682542Z","shell.execute_reply":"2025-09-16T06:36:50.687669Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !pip install monai","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:36:50.69304Z","iopub.execute_input":"2025-09-16T06:36:50.693346Z","iopub.status.idle":"2025-09-16T06:36:50.706956Z","shell.execute_reply.started":"2025-09-16T06:36:50.693322Z","shell.execute_reply":"2025-09-16T06:36:50.705975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!python -m pip install iterative-stratification","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:36:50.708481Z","iopub.execute_input":"2025-09-16T06:36:50.709502Z","iopub.status.idle":"2025-09-16T06:36:56.074177Z","shell.execute_reply.started":"2025-09-16T06:36:50.709467Z","shell.execute_reply":"2025-09-16T06:36:56.072986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom pathlib import Path\nfrom tqdm import tqdm\nimport copy\nimport ast\nfrom collections import defaultdict, Counter\nfrom glob import glob\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport nibabel as nib\nfrom skmultilearn.model_selection import IterativeStratification\nimport cv2\nfrom scipy.ndimage import zoom\nimport pydicom\nfrom iterstrat.ml_stratifiers import MultilabelStratifiedKFold","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:36:56.076019Z","iopub.execute_input":"2025-09-16T06:36:56.076486Z","iopub.status.idle":"2025-09-16T06:36:58.490093Z","shell.execute_reply.started":"2025-09-16T06:36:56.076453Z","shell.execute_reply":"2025-09-16T06:36:58.489122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/rsna-intracranial-aneurysm-detection/train.csv')\ndf_loc = pd.read_csv('/kaggle/input/rsna-intracranial-aneurysm-detection/train_localizers.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:36:58.491085Z","iopub.execute_input":"2025-09-16T06:36:58.49162Z","iopub.status.idle":"2025-09-16T06:36:58.560284Z","shell.execute_reply.started":"2025-09-16T06:36:58.491566Z","shell.execute_reply":"2025-09-16T06:36:58.55937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Aneurysm Present'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:36:58.562022Z","iopub.execute_input":"2025-09-16T06:36:58.56247Z","iopub.status.idle":"2025-09-16T06:36:58.580048Z","shell.execute_reply.started":"2025-09-16T06:36:58.562438Z","shell.execute_reply":"2025-09-16T06:36:58.57908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_loc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:36:58.581117Z","iopub.execute_input":"2025-09-16T06:36:58.581485Z","iopub.status.idle":"2025-09-16T06:36:58.617458Z","shell.execute_reply.started":"2025-09-16T06:36:58.581459Z","shell.execute_reply":"2025-09-16T06:36:58.616298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_loc['SeriesInstanceUID'].iloc[0], df_loc['SOPInstanceUID'].iloc[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:36:58.618459Z","iopub.execute_input":"2025-09-16T06:36:58.619371Z","iopub.status.idle":"2025-09-16T06:36:58.626101Z","shell.execute_reply.started":"2025-09-16T06:36:58.619329Z","shell.execute_reply":"2025-09-16T06:36:58.625129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def norm_img(img):\n    vmin = np.min(img)\n    vmax = np.max(img)\n    img = (img - vmin) / (vmax - vmin)\n    img = (img*255).astype(np.uint8)\n    return img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:36:58.627167Z","iopub.execute_input":"2025-09-16T06:36:58.628599Z","iopub.status.idle":"2025-09-16T06:36:58.642303Z","shell.execute_reply.started":"2025-09-16T06:36:58.628541Z","shell.execute_reply":"2025-09-16T06:36:58.64123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import Counter\nfrom typing import List, Tuple, Optional, Dict\nimport os\nimport numpy as np\nimport pydicom\nfrom pydicom.multival import MultiValue\nfrom pydicom.valuerep import DSfloat\n\n\nclass DicomVolumeFiles:\n    \"\"\"\n    目的：\n      - 2D：旧版と完全一致のメタデータ（= トップレベルのみ、補完・正規化なし）\n      - 3D（Enhanced / Multi-frame, PFFGSあり前提だがIOPの所在は多様）：\n          * IOPの所在を Shared → PerFrame走査 → Top-level → 無ければ IPP から推定\n          * slice position = dot( normal(IOP), IPP ) の昇順でフレームを並び替え\n          * file_paths は 3D では 1 件のみ\n          * 実効スケールは PixelSpacingEffective / SliceThicknessEffective / SpacingBetweenSlicesEffective\n          * TAGS は旧版互換（トップレベルだけコピー）。ただし 3D でトップレベル欠損時は\n            PFFGS/Shared から PixelSpacing / SliceThickness / SpacingBetweenSlices を\n            “欠損項目に限って” 補完（最頻値など）する\n    \"\"\"\n\n    TAGS = [\n        'BitsAllocated',\n        'BitsStored',\n        'Columns',\n        'FrameOfReferenceUID',\n        'HighBit',\n        'ImageOrientationPatient',\n        'ImagePositionPatient',\n        'InstanceNumber',\n        'Modality',\n        'PatientID',\n        'PhotometricInterpretation',\n        'PixelRepresentation',\n        'PixelSpacing',\n        'PlanarConfiguration',\n        'RescaleIntercept',\n        'RescaleSlope',\n        'RescaleType',\n        'Rows',\n        'SOPClassUID',\n        'SOPInstanceUID',\n        'SamplesPerPixel',\n        'SliceThickness',\n        'SpacingBetweenSlices',\n        'StudyInstanceUID',\n        'TransferSyntaxUID'\n    ]\n\n    def __init__(self, directory: str, plane: str = \"auto\", debug: bool = False):\n        self.plane = plane.lower()\n        assert self.plane in {\"axial\", \"sagittal\", \"coronal\", \"auto\"}\n        self.debug = debug\n\n        # 3D/マルチフレーム関連（補助情報）\n        self.is_multiframe: bool = False\n        self.n_frames: Optional[int] = None\n        self.frame_indices: Optional[List[int]] = None\n        self.read_order_like_2d: Optional[List[Tuple[str, int]]] = None\n        self.order_info: Optional[Dict] = None\n\n        # 共通\n        self.file_paths: List[str] = []\n        self.slice_positions: List[float] = []\n\n        # 実効スケール（互換維持のため TAGS は上書きしない）\n        self.PixelSpacingEffective: Optional[Tuple[float, float]] = None\n        self.SliceThicknessEffective: Optional[float] = None\n        self.SpacingBetweenSlicesEffective: Optional[float] = None\n\n        self._load_dicom_series(directory)\n\n    # ---------- 2D 従来の面推定・スライスキー（トップレベルのみ使用） ----------\n    def _slice_key(self, ds: pydicom.Dataset) -> float:\n        ipp = np.array(list(map(float, ds.ImagePositionPatient)))\n        if self.plane == \"axial\":\n            return ipp[2]\n        elif self.plane == \"sagittal\":\n            return ipp[0]\n        elif self.plane == \"coronal\":\n            return ipp[1]\n        else:\n            rx, ry, rz, cx, cy, cz = map(float, ds.ImageOrientationPatient)\n            n = np.cross([rx, ry, rz], [cx, cy, cz])\n            return float(np.dot(ipp, n))\n\n    def _infer_plane(self, ds: pydicom.Dataset) -> str:\n        rx, ry, rz, cx, cy, cz = map(float, ds.ImageOrientationPatient)\n        n = np.cross([rx, ry, rz], [cx, cy, cz])\n        axis = int(np.argmax(np.abs(n)))  # 0:X,1:Y,2:Z\n        return [\"sagittal\", \"coronal\", \"axial\"][axis]\n\n    def _infer_plane_from_normal(self, n: np.ndarray) -> str:\n        axis = int(np.argmax(np.abs(n)))  # 0:X,1:Y,2:Z\n        return [\"sagittal\", \"coronal\", \"axial\"][axis]\n\n    # ---------- 3D 用ヘルパ（属性へは書き込まない：内部計算専用） ----------\n    @staticmethod\n    def _normalize_vec(v: np.ndarray) -> Optional[np.ndarray]:\n        nrm = float(np.linalg.norm(v))\n        if nrm == 0.0 or not np.isfinite(nrm):\n            return None\n        return v / nrm\n\n    def _get_iop_from_shared(self, ds: pydicom.Dataset) -> Optional[List[float]]:\n        if 'SharedFunctionalGroupsSequence' in ds:\n            sfg = ds.SharedFunctionalGroupsSequence[0]\n            if 'PlaneOrientationSequence' in sfg:\n                return [float(x) for x in sfg.PlaneOrientationSequence[0].ImageOrientationPatient]\n        return None\n\n    def _get_iop_from_pffgs(self, ds: pydicom.Dataset) -> Optional[List[float]]:\n        if 'PerFrameFunctionalGroupsSequence' in ds:\n            for fgs in ds.PerFrameFunctionalGroupsSequence:\n                if 'PlaneOrientationSequence' in fgs:\n                    return [float(x) for x in fgs.PlaneOrientationSequence[0].ImageOrientationPatient]\n        return None\n\n    def _get_positions_from_pffgs(self, ds: pydicom.Dataset) -> List[Optional[np.ndarray]]:\n        pos: List[Optional[np.ndarray]] = []\n        if 'PerFrameFunctionalGroupsSequence' in ds:\n            for fgs in ds.PerFrameFunctionalGroupsSequence:\n                if 'PlanePositionSequence' in fgs:\n                    ipp = fgs.PlanePositionSequence[0].ImagePositionPatient\n                    pos.append(np.array([float(v) for v in ipp], dtype=float))\n                else:\n                    pos.append(None)\n        return pos\n\n    def _estimate_normal_from_positions(self, pos_list: List[Optional[np.ndarray]]) -> Optional[np.ndarray]:\n        pts = np.array([p for p in pos_list if p is not None], dtype=float)\n        if pts.shape[0] < 2:\n            return None\n        # PCA の第一主成分 or 端点差で方向推定\n        pts_c = pts - pts.mean(axis=0, keepdims=True)\n        try:\n            _, _, vh = np.linalg.svd(pts_c, full_matrices=False)\n            axis = vh[0]  # 最大分散方向\n        except Exception:\n            axis = pts[-1] - pts[0]\n        return self._normalize_vec(axis)\n\n    def _pick_normal(self, ds: pydicom.Dataset, pos_list: List[Optional[np.ndarray]]) -> Tuple[np.ndarray, str]:\n        \"\"\"\n        IOP 正規化法線 n と、その取得ソースを返す。\n        優先度: Shared → PerFrame走査 → Top-level → IPP推定\n        \"\"\"\n        # 1) Shared\n        iop = self._get_iop_from_shared(ds)\n        if iop is not None:\n            r = np.array(iop[:3], dtype=float); c = np.array(iop[3:], dtype=float)\n            n = self._normalize_vec(np.cross(r, c))\n            if n is not None:\n                return n, \"shared\"\n\n        # 2) PerFrame 走査\n        iop = self._get_iop_from_pffgs(ds)\n        if iop is not None:\n            r = np.array(iop[:3], dtype=float); c = np.array(iop[3:], dtype=float)\n            n = self._normalize_vec(np.cross(r, c))\n            if n is not None:\n                return n, \"per-frame\"\n\n        # 3) Top-level\n        if 'ImageOrientationPatient' in ds:\n            iop = [float(x) for x in ds.ImageOrientationPatient]\n            r = np.array(iop[:3], dtype=float); c = np.array(iop[3:], dtype=float)\n            n = self._normalize_vec(np.cross(r, c))\n            if n is not None:\n                return n, \"top-level\"\n\n        # 4) フォールバック：IPP から推定\n        n = self._estimate_normal_from_positions(pos_list)\n        if n is not None:\n            return n, \"ipp-pca\"\n\n        raise ValueError(\"IOP（向き）が取得・推定できませんでした。\")\n\n    def _pixel_measures_effective(self, ds: pydicom.Dataset) -> Tuple[Optional[Tuple[float, float]], Optional[float], Optional[float]]:\n        \"\"\"\n        実効 PixelSpacing / SliceThickness / SpacingBetweenSlices を返す。\n        優先度: Shared PixelMeasures → PerFrame(先頭で見つかったもの)\n        \"\"\"\n        dy = dx = th = sb = None\n        # Shared\n        if 'SharedFunctionalGroupsSequence' in ds:\n            sfg = ds.SharedFunctionalGroupsSequence[0]\n            if 'PixelMeasuresSequence' in sfg:\n                pm = sfg.PixelMeasuresSequence[0]\n                if 'PixelSpacing' in pm:\n                    dy, dx = [float(v) for v in pm.PixelSpacing]\n                if 'SliceThickness' in pm:\n                    th = float(pm.SliceThickness)\n                if 'SpacingBetweenSlices' in pm:\n                    sb = float(pm.SpacingBetweenSlices)\n        # PerFrame（見つからなかった項目だけ補完）\n        if (dy is None or dx is None or th is None or sb is None) and \\\n           'PerFrameFunctionalGroupsSequence' in ds:\n            for fgs in ds.PerFrameFunctionalGroupsSequence:\n                if 'PixelMeasuresSequence' in fgs:\n                    pm = fgs.PixelMeasuresSequence[0]\n                    if (dy is None or dx is None) and 'PixelSpacing' in pm:\n                        dy, dx = [float(v) for v in pm.PixelSpacing]\n                    if th is None and 'SliceThickness' in pm:\n                        th = float(pm.SliceThickness)\n                    if sb is None and 'SpacingBetweenSlices' in pm:\n                        sb = float(pm.SpacingBetweenSlices)\n                    if dy is not None and dx is not None and th is not None and sb is not None:\n                        break\n        return ((dy, dx) if (dy is not None and dx is not None) else None, th, sb)\n\n    def _backfill_spacing_tags_from_fgs(self, ds: pydicom.Dataset):\n        \"\"\"\n        3D向け：トップレベルの PixelSpacing/SliceThickness/SpacingBetweenSlices が欠損なら、\n        Shared/PFFGS から取得して TAGS に“欠損項目のみ”補完する（2D互換は崩さない）。\n        - PixelSpacing がフレームごとに変化する場合は、丸め(6桁)後の最頻値を代表値として採用。\n        \"\"\"\n        need_ps = getattr(self, 'PixelSpacing', None) is None\n        need_th = getattr(self, 'SliceThickness', None) is None\n        need_sb = getattr(self, 'SpacingBetweenSlices', None) is None\n        if not (need_ps or need_th or need_sb):\n            return\n\n        # 既存の実効値推定\n        ps_eff, th_eff, sb_eff = self._pixel_measures_effective(ds)\n\n        # PFFGS 全フレームの候補を集計\n        ps_list = []\n        th_list = []\n        sb_list = []\n        if 'PerFrameFunctionalGroupsSequence' in ds:\n            for fgs in ds.PerFrameFunctionalGroupsSequence:\n                if 'PixelMeasuresSequence' in fgs:\n                    pm = fgs.PixelMeasuresSequence[0]\n                    if 'PixelSpacing' in pm:\n                        dy, dx = [float(v) for v in pm.PixelSpacing]\n                        ps_list.append((dy, dx))\n                    if 'SliceThickness' in pm:\n                        th_list.append(float(pm.SliceThickness))\n                    if 'SpacingBetweenSlices' in pm:\n                        sb_list.append(float(pm.SpacingBetweenSlices))\n\n        # 代表値の選び方\n        def _mode_pair(pairs):\n            if not pairs:\n                return None\n            def key(p): return (round(p[0], 6), round(p[1], 6))\n            from collections import Counter as C2\n            k, _ = C2(key(p) for p in pairs).most_common(1)[0]\n            return (float(k[0]), float(k[1]))\n\n        def _mode_scalar(vals):\n            if not vals:\n                return None\n            from collections import Counter as C2\n            k, _ = C2(round(x, 6) for x in vals).most_common(1)[0]\n            return float(k)\n\n        # 欠損項目を優先順で埋める\n        if need_ps:\n            val = None\n            # Shared/PerFrame先頭（＝ps_eff）→ 全体モード → ps_eff（保険）\n            if ps_eff is not None:\n                val = ps_eff\n            m = _mode_pair(ps_list)\n            if m is not None:\n                val = m\n            if val is not None:\n                dy, dx = val\n                try:\n                    self.PixelSpacing = MultiValue(DSfloat, [dy, dx])\n                except Exception:\n                    self.PixelSpacing = [dy, dx]\n\n        if need_th:\n            val = th_eff if th_eff is not None else _mode_scalar(th_list)\n            if val is not None:\n                try:\n                    self.SliceThickness = DSfloat(val)\n                except Exception:\n                    self.SliceThickness = val\n\n        if need_sb:\n            val = sb_eff if sb_eff is not None else _mode_scalar(sb_list)\n            if val is not None:\n                try:\n                    self.SpacingBetweenSlices = DSfloat(val)\n                except Exception:\n                    self.SpacingBetweenSlices = val\n\n    # ---------- ロード本体 ----------\n    def _load_dicom_series(self, directory: str):\n        paths = [\n            os.path.join(directory, fn)\n            for fn in os.listdir(directory)\n            if os.path.isfile(os.path.join(directory, fn))\n        ]\n        if not paths:\n            raise ValueError(\"有効な DICOM ファイルが見つかりません。\")\n\n        sizes = []\n        multiframes = []\n        for p in paths:\n            try:\n                ds = pydicom.dcmread(p, stop_before_pixels=True)\n                sizes.append((int(ds.Rows), int(ds.Columns)))\n                if int(ds.get('NumberOfFrames', 1)) > 1 or 'PerFrameFunctionalGroupsSequence' in ds:\n                    multiframes.append((p, ds))\n            except Exception:\n                continue\n\n        if self.debug:\n            print('Num images:', len(paths))\n        if not sizes:\n            raise ValueError(\"Rows/Columns を取得できませんでした。\")\n\n        # ---- 3D（Enhanced / Multi-frame）経路 ----\n        if multiframes:\n            path, ds = multiframes[0]\n            self.is_multiframe = True\n\n            if 'PerFrameFunctionalGroupsSequence' not in ds:\n                raise ValueError(\"3D想定ですが PerFrameFunctionalGroupsSequence がありません。\")\n            pffgs = ds.PerFrameFunctionalGroupsSequence\n            self.n_frames = int(ds.get('NumberOfFrames', len(pffgs)))\n\n            # 位置（IPP）取得\n            pos_list = self._get_positions_from_pffgs(ds)\n            if not pos_list or all(p is None for p in pos_list):\n                raise ValueError(\"PerFrame の IPP（PlanePositionSequence）が取得できませんでした。\")\n\n            # 法線 n を取得（Shared → PerFrame走査 → Top-level → IPP推定）\n            n, n_source = self._pick_normal(ds, pos_list)\n\n            # slice position スカラー算出\n            scalars: List[float] = []\n            for p0 in pos_list:\n                scalars.append(float(np.dot(n, p0)) if p0 is not None else np.nan)\n\n            # 安定ソート：slice position 昇順（同値は元フレーム番号でタイブレーク）\n            idx_sorted = sorted(\n                range(len(scalars)),\n                key=lambda i: (np.nan_to_num(scalars[i], nan=np.inf), i)\n            )\n            # NaN は末尾へ\n            idx_sorted = [i for i in idx_sorted if not np.isnan(scalars[i])] + \\\n                         [i for i in idx_sorted if np.isnan(scalars[i])]\n\n            # 3D は file_paths を 1 件のみ保持\n            self.file_paths = [path]\n            self.frame_indices = idx_sorted\n            self.slice_positions = [scalars[i] for i in idx_sorted]\n            self.read_order_like_2d = [(path, i) for i in idx_sorted]\n\n            # plane='auto' は内部法線から決定（TAGSへは書かない）\n            if self.plane == \"auto\":\n                self.plane = self._infer_plane_from_normal(n)\n                if self.debug:\n                    print(f\"[auto] inferred plane = {self.plane} (source={n_source})\")\n\n            # 並び替え情報のメモ\n            self.order_info = {\n                \"method\": \"dot(normal(IOP or IPP-fit), IPP)\",\n                \"direction\": \"ascending\",\n                \"orientation_source\": n_source,   # 'shared'/'per-frame'/'top-level'/'ipp-pca'\n                \"frames\": len(idx_sorted),\n            }\n\n            # 実効スケール（TAGSは上書きしない）\n            ps_eff, th_eff, sb_eff = self._pixel_measures_effective(ds)\n            self.PixelSpacingEffective = ps_eff\n            self.SliceThicknessEffective = th_eff\n            self.SpacingBetweenSlicesEffective = sb_eff\n\n            # メタデータ（TAGS）は旧版完全一致：トップレベルだけをコピー（ピクセルも読む）\n            self._set_metadata_top_level_only(pydicom.dcmread(path))\n\n            # ★3Dのときに限り、トップレベル欠損項目は PFFGS/Shared から TAGS を補完\n            self._backfill_spacing_tags_from_fgs(ds)\n            return\n\n        # ---- 2D（従来スタック）経路：旧版完全一致 ----\n        mode_size, _ = Counter(sizes).most_common(1)[0]\n        if self.debug:\n            print(\"Chosen image size (Rows, Columns):\", mode_size)\n\n        dicom_files = []\n        for p in paths:\n            try:\n                ds = pydicom.dcmread(p, stop_before_pixels=True)\n\n                if self.plane == \"auto\":\n                    self.plane = self._infer_plane(ds)\n                    if self.debug:\n                        print(f\"[auto] inferred plane = {self.plane}\")\n\n                if (int(ds.Rows), int(ds.Columns)) != mode_size:\n                    print(\"Rows/Columnsと画像サイズが一致しません：\", int(ds.Rows), mode_size)\n                else:\n                    z = self._slice_key(ds) if \"ImagePositionPatient\" in ds else 0.0\n                    dicom_files.append((z, p))\n            except Exception:\n                continue\n\n        if not dicom_files:\n            raise ValueError(\"モードの Rows/Columns と一致する DICOM がありません。\")\n\n        dicom_files.sort(key=lambda t: t[0])\n        self.file_paths = [path for z, path in dicom_files]   # 2Dはファイル数ぶん\n        self.slice_positions = [z for z, path in dicom_files]\n        self.n_frames = None\n        self.frame_indices = None\n        self.read_order_like_2d = None\n\n        # 旧版と同じ：最初の1枚はピクセルも読む\n        ds0 = pydicom.dcmread(self.file_paths[0])\n        self._set_metadata_top_level_only(ds0)\n\n        # 実効スケール（2Dはトップレベルをそのまま）\n        if hasattr(ds0, 'PixelSpacing'):\n            try:\n                dy, dx = [float(v) for v in ds0.PixelSpacing]\n                self.PixelSpacingEffective = (dy, dx)\n            except Exception:\n                self.PixelSpacingEffective = None\n        self.SliceThicknessEffective = float(ds0.SliceThickness) if hasattr(ds0, 'SliceThickness') else None\n        self.SpacingBetweenSlicesEffective = float(ds0.SpacingBetweenSlices) if hasattr(ds0, 'SpacingBetweenSlices') else None\n\n    # ---------- メタデータ設定（旧版完全一致） ----------\n    def _set_metadata_top_level_only(self, ds: pydicom.Dataset):\n        \"\"\"トップレベルに存在する値だけをそのまま代入（補完・正規化なし）\"\"\"\n        for tag in self.TAGS:\n            setattr(self, tag, getattr(ds, tag, None))\n\n    # ---------- 公開プロパティ ----------\n    @property\n    def z_spacing(self) -> Optional[float]:\n        \"\"\"\n        隣接スライス間の平均間隔（mm）を返す（旧版と同じ計算）。\n        2D: self.slice_positions は各ファイルのスカラー位置\n        3D: self.slice_positions は各フレームのスカラー位置\n        \"\"\"\n        if not self.slice_positions or len(self.slice_positions) < 2:\n            return None\n        diffs = np.diff(sorted(self.slice_positions))\n        return float(np.mean(np.abs(diffs)))\n\n    # ---------- 補助：2D互換の読み出し計画 ----------\n    def get_ordered_read_plan(self) -> List[Tuple[str, Optional[int]]]:\n        \"\"\"\n        2D: [(path, None), ...]\n        3D: [(path, frame_index), ...]\n        \"\"\"\n        if self.is_multiframe:\n            if self.read_order_like_2d is not None:\n                return list(self.read_order_like_2d)\n            if self.file_paths and self.frame_indices:\n                return [(self.file_paths[0], i) for i in self.frame_indices]\n            return []\n        else:\n            return [(p, None) for p in self.file_paths]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:36:58.643365Z","iopub.execute_input":"2025-09-16T06:36:58.643648Z","iopub.status.idle":"2025-09-16T06:36:58.700405Z","shell.execute_reply.started":"2025-09-16T06:36:58.643623Z","shell.execute_reply":"2025-09-16T06:36:58.699541Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dcm = DicomVolumeFiles('/kaggle/input/rsna-intracranial-aneurysm-detection/series/1.2.826.0.1.3680043.8.498.10004044428023505108375152878107656647')\nlen(dcm.file_paths)\nsp = dcm.slice_positions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:36:58.701514Z","iopub.execute_input":"2025-09-16T06:36:58.701846Z","iopub.status.idle":"2025-09-16T06:37:00.24733Z","shell.execute_reply.started":"2025-09-16T06:36:58.701821Z","shell.execute_reply":"2025-09-16T06:37:00.246265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dcm.SpacingBetweenSlices","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:37:00.250523Z","iopub.execute_input":"2025-09-16T06:37:00.250839Z","iopub.status.idle":"2025-09-16T06:37:00.256938Z","shell.execute_reply.started":"2025-09-16T06:37:00.250815Z","shell.execute_reply":"2025-09-16T06:37:00.256112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.diff(sp), dcm.plane","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:37:00.257755Z","iopub.execute_input":"2025-09-16T06:37:00.258091Z","iopub.status.idle":"2025-09-16T06:37:00.279943Z","shell.execute_reply.started":"2025-09-16T06:37:00.258063Z","shell.execute_reply":"2025-09-16T06:37:00.279044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = list(Path('/kaggle/input/rsna-intracranial-aneurysm-detection/series/1.2.826.0.1.3680043.8.498.10004044428023505108375152878107656647').glob('*'))\nlen(path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:37:00.280952Z","iopub.execute_input":"2025-09-16T06:37:00.281207Z","iopub.status.idle":"2025-09-16T06:37:00.299507Z","shell.execute_reply.started":"2025-09-16T06:37:00.281187Z","shell.execute_reply":"2025-09-16T06:37:00.29854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dcm.PixelSpacing, dcm.z_spacing","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:37:00.300475Z","iopub.execute_input":"2025-09-16T06:37:00.300781Z","iopub.status.idle":"2025-09-16T06:37:00.31543Z","shell.execute_reply.started":"2025-09-16T06:37:00.300753Z","shell.execute_reply":"2025-09-16T06:37:00.314489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# eda_path = '/kaggle/input/rsna2025-extra/train_eda.csv'\n# FORCE = True\n# if (not os.path.exists(eda_path)) | FORCE:\n#     series_root = Path('/kaggle/input/rsna-intracranial-aneurysm-detection/series')\n#     rows = []\n#     for _, row in tqdm(df.iterrows(), total=len(df)):\n#         series_name = row['SeriesInstanceUID']\n#         series_path = series_root / series_name\n#         img_paths = list(series_path.glob('*'))\n#         num_slices = len(img_paths)\n#         img_path = img_paths[0]   # そのシリーズの 1 枚目\n#         try:\n#             dcm = pydicom.dcmread(img_path)\n#             if hasattr(dcm, 'AngioFlag'):\n#                 angio = dcm.AngioFlag\n#             else:\n#                 angio = 'UK'\n    \n#             # ------- 1) 画像サイズ関連 -------\n#             shape = dcm.pixel_array.shape\n#             if len(shape) == 2:                       # 2D\n#                 y, x = shape\n#                 z = 1\n#             else:                                     # 3D（Volume）\n#                 z, y, x = shape\n    \n#             # ------- 2) Allowlist タグ -------\n#             row = {\n#                 \"SeriesInstanceUID\": series_name,\n#                 \"x\": x,\n#                 \"y\": y,\n#                 \"z\": num_slices,          # 実際のスライス枚数\n#                 \"ndim\": len(shape),\n#                 \"angio\": angio,\n#                 \"z_spacing\": \n#             }\n#             for tag in DICOM_TAG_ALLOWLIST:\n#                 value = getattr(dcm, tag, None)\n#                 # シーケンスや MultiValue はそのままでは入らないので文字列化\n#                 if isinstance(value, (list, pydicom.multival.MultiValue)):\n#                     value = \",\".join(map(str, value))\n#                 row[tag] = value\n    \n#             # 例：よく使う追加タグ\n#             row[\"Modality\"]      = getattr(dcm, \"Modality\", None)\n#             row[\"PixelSpacing\"]  = \",\".join(map(str, getattr(dcm, \"PixelSpacing\", [])))\n    \n#             rows.append(row)\n    \n#         except Exception as e:\n#             print(f\"読込失敗: {img_path} — {e}\")\n    \n#     # ---------- DataFrame 化 ----------\n#     df_eda = pd.DataFrame(rows)\n# else:\n#     df_eda = pd.read_csv(eda_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:37:00.316381Z","iopub.execute_input":"2025-09-16T06:37:00.31668Z","iopub.status.idle":"2025-09-16T06:37:00.3344Z","shell.execute_reply.started":"2025-09-16T06:37:00.316656Z","shell.execute_reply":"2025-09-16T06:37:00.333379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df = df[:100]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:37:00.335512Z","iopub.execute_input":"2025-09-16T06:37:00.335813Z","iopub.status.idle":"2025-09-16T06:37:00.357018Z","shell.execute_reply.started":"2025-09-16T06:37:00.335791Z","shell.execute_reply":"2025-09-16T06:37:00.356012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------------------------------------------------\n# EDA table creation with DicomVolumeFiles\n# ------------------------------------------------------------\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nimport os, pydicom, pandas as pd\n\neda_path = '/kaggle/input/rsna2025-extra/train_eda.csv'\nseries_root = Path('/kaggle/input/rsna-intracranial-aneurysm-detection/series')\nFORCE = True      # 上書きするとき True\n\nif FORCE or not os.path.exists(eda_path):\n    rows = []\n\n    for _, row in tqdm(df.iterrows(), total=len(df)):\n        sid         = row['SeriesInstanceUID']\n        series_dir  = series_root / sid\n\n        try:\n            vol      = DicomVolumeFiles(str(series_dir))      # ★ クラス利用\n            num_slices = len(vol.file_paths)\n            first_ds   = pydicom.dcmread(vol.file_paths[0], stop_before_pixels=True)\n    \n            # --- 基本情報 ---\n            if num_slices==1:\n                ndim = 3\n            else:\n                ndim =  2\n            angio = getattr(first_ds, 'AngioFlag', 'UK')      # 無ければ 'UK'\n    \n            row_dict = {\n                \"SeriesInstanceUID\": sid,\n                \"num_slice\"      : num_slices,\n                \"ndim\"   : ndim,\n                \"angio\"  : angio,\n                \"z_spacing\": vol.z_spacing,\n                \"sorted_files\": vol.file_paths,\n                \"slice_positions\": vol.slice_positions\n            }\n    \n            # --- TAGS 内メタデータを追加 ---\n            for tag in DicomVolumeFiles.TAGS:\n                val = getattr(vol, tag, None)\n                if isinstance(val, (list, pydicom.multival.MultiValue)):\n                    val = \",\".join(map(str, val))\n                row_dict[tag] = val\n    \n            rows.append(row_dict)\n\n        except Exception as e:\n            print(f\"[WARN] Series {sid} でエラー: {e}\")\n\n    df_eda = pd.DataFrame(rows)\n    #df_eda.to_csv(eda_path, index=False)\n    print(f\"✓ Saved EDA table to {eda_path}\")\n\nelse:\n    df_eda = pd.read_csv(eda_path)\n    print(f\"✓ Loaded cached EDA table from {eda_path}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:37:00.358192Z","iopub.execute_input":"2025-09-16T06:37:00.358454Z","iopub.status.idle":"2025-09-16T06:39:38.700419Z","shell.execute_reply.started":"2025-09-16T06:37:00.358432Z","shell.execute_reply":"2025-09-16T06:39:38.699415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_eda.iloc[4]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:38.701378Z","iopub.execute_input":"2025-09-16T06:39:38.701661Z","iopub.status.idle":"2025-09-16T06:39:38.710066Z","shell.execute_reply.started":"2025-09-16T06:39:38.701638Z","shell.execute_reply":"2025-09-16T06:39:38.709014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_eda['z_spacing'].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:38.71096Z","iopub.execute_input":"2025-09-16T06:39:38.711282Z","iopub.status.idle":"2025-09-16T06:39:38.730156Z","shell.execute_reply.started":"2025-09-16T06:39:38.711259Z","shell.execute_reply":"2025-09-16T06:39:38.729178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.merge(df_eda, how=\"left\", on='SeriesInstanceUID', suffixes=(\"\", \"_eda\") )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:38.731192Z","iopub.execute_input":"2025-09-16T06:39:38.731467Z","iopub.status.idle":"2025-09-16T06:39:38.757311Z","shell.execute_reply.started":"2025-09-16T06:39:38.731448Z","shell.execute_reply":"2025-09-16T06:39:38.756211Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['sorted_files']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:38.758432Z","iopub.execute_input":"2025-09-16T06:39:38.75879Z","iopub.status.idle":"2025-09-16T06:39:38.784162Z","shell.execute_reply.started":"2025-09-16T06:39:38.758761Z","shell.execute_reply":"2025-09-16T06:39:38.783412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sid = df['SeriesInstanceUID'].iloc[4]\npath = list(Path(f'/kaggle/input/rsna-intracranial-aneurysm-detection/series/{sid}').glob('*'))[0]\ndcm = pydicom.dcmread(path)\nper_frame = dcm.PerFrameFunctionalGroupsSequence\npixel_spacings = []\nfor i, fg in enumerate(per_frame):\n    if \"PixelMeasuresSequence\" in fg:\n        px = fg.PixelMeasuresSequence[0].PixelSpacing  # [row, col] in mm\n        pixel_spacings.append([float(px[0]), float(px[1])])\n    else:\n        pixel_spacings.append(None)  # そのフレームに未記載\nif 'SharedFunctionalGroupsSequence' in dcm:\n    sfg = dcm.SharedFunctionalGroupsSequence[0]\n    if 'PixelMeasuresSequence' in sfg:\n        pm = sfg.PixelMeasuresSequence[0]\n        if 'PixelSpacing' in pm:\n            dy, dx = [float(v) for v in pm.PixelSpacing]\n        if 'SliceThickness' in pm:\n            th = float(pm.SliceThickness)\n        if 'SpacingBetweenSlices' in pm:\n            sb = float(pm.SpacingBetweenSlices)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T07:36:18.112505Z","iopub.execute_input":"2025-09-16T07:36:18.112899Z","iopub.status.idle":"2025-09-16T07:36:18.141967Z","shell.execute_reply.started":"2025-09-16T07:36:18.112875Z","shell.execute_reply":"2025-09-16T07:36:18.141062Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dy, dx","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T07:36:33.301407Z","iopub.execute_input":"2025-09-16T07:36:33.301747Z","iopub.status.idle":"2025-09-16T07:36:33.308084Z","shell.execute_reply.started":"2025-09-16T07:36:33.301722Z","shell.execute_reply":"2025-09-16T07:36:33.307125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:38.785277Z","iopub.execute_input":"2025-09-16T06:39:38.785643Z","iopub.status.idle":"2025-09-16T06:39:38.803234Z","shell.execute_reply.started":"2025-09-16T06:39:38.785606Z","shell.execute_reply":"2025-09-16T06:39:38.802346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['slice_positions']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:38.804036Z","iopub.execute_input":"2025-09-16T06:39:38.804535Z","iopub.status.idle":"2025-09-16T06:39:38.827183Z","shell.execute_reply.started":"2025-09-16T06:39:38.804494Z","shell.execute_reply":"2025-09-16T06:39:38.826235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_eda['angio'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:38.828115Z","iopub.execute_input":"2025-09-16T06:39:38.828448Z","iopub.status.idle":"2025-09-16T06:39:38.84947Z","shell.execute_reply.started":"2025-09-16T06:39:38.828423Z","shell.execute_reply":"2025-09-16T06:39:38.848224Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"root_seg = Path('/kaggle/input/rsna-intracranial-aneurysm-detection/segmentations')\npaths_seg = root_seg.glob('*')\nids_seg = [path_seg.name for path_seg in paths_seg]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:42.716785Z","iopub.execute_input":"2025-09-16T06:39:42.717077Z","iopub.status.idle":"2025-09-16T06:39:42.785361Z","shell.execute_reply.started":"2025-09-16T06:39:42.717055Z","shell.execute_reply":"2025-09-16T06:39:42.784406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Modality'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:42.78664Z","iopub.execute_input":"2025-09-16T06:39:42.78701Z","iopub.status.idle":"2025-09-16T06:39:42.795334Z","shell.execute_reply.started":"2025-09-16T06:39:42.786987Z","shell.execute_reply":"2025-09-16T06:39:42.794335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─── 0=背景, 1‑13=任意の色 ───\nPALETTE = np.array([\n    [  0,   0,   0],   # 0: 背景 (黒)\n    [230,  25,  75],   # 1: 赤系\n    [ 60, 180,  75],   # 2: 緑\n    [255, 225,  25],   # 3: 黄\n    [  0, 130, 200],   # 4: 青\n    [245, 130,  48],   # 5: オレンジ\n    [145,  30, 180],   # 6: 紫\n    [ 70, 240, 240],   # 7: 水色\n    [240,  50, 230],   # 8: ピンク\n    [210, 245,  60],   # 9: 黄緑\n    [250, 190, 190],   #10: 肌色\n    [  0, 128, 128],   #11: 青緑\n    [230, 190, 255],   #12: 薄紫\n    [170, 110,  40],   #13: 茶\n], dtype=np.uint8)\n\ndef mask_to_rgb(mask: np.ndarray, palette: np.ndarray = PALETTE) -> np.ndarray:\n    \"\"\"\n    uint8 2D ラベルマスクを RGB 配色画像に変換する。\n    \n    Parameters\n    ----------\n    mask : (H, W) ndarray of uint8\n        0 = 背景, 1‑13 = ラベル ID\n    palette : (N, 3) ndarray of uint8\n        ラベル ID → (R,G,B) の対応表。先頭行は背景色。\n    \n    Returns\n    -------\n    rgb : (H, W, 3) ndarray of uint8\n        カラーマスク。OpenCV や matplotlib でそのまま表示可能。\n    \"\"\"\n    if mask.dtype != np.uint8:\n        raise ValueError(\"mask は uint8 である必要があります\")\n    if mask.max() >= len(palette):\n        raise ValueError(f\"mask に {len(palette)-1} を超えるラベルが含まれています\")\n\n    # 高速: palette をインデックスとして直接ブロードキャスト\n    return palette[mask]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:42.796308Z","iopub.execute_input":"2025-09-16T06:39:42.796605Z","iopub.status.idle":"2025-09-16T06:39:42.812346Z","shell.execute_reply.started":"2025-09-16T06:39:42.796552Z","shell.execute_reply":"2025-09-16T06:39:42.811253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"series_name = df_eda['SeriesInstanceUID'].iloc[0].split('/')[-1]\nseries_name = '1.2.826.0.1.3680043.8.498.10076056930521523789588901704956188485'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:42.813313Z","iopub.execute_input":"2025-09-16T06:39:42.81366Z","iopub.status.idle":"2025-09-16T06:39:42.836627Z","shell.execute_reply.started":"2025-09-16T06:39:42.81363Z","shell.execute_reply":"2025-09-16T06:39:42.835379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# segmentation_series = list(Path('/kaggle/input/rsna-intracranial-aneurysm-detection/segmentations').glob('*'))\n# segmentation_series = [path.name for path in segmentation_series]\n# segmentation_series","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:42.838084Z","iopub.execute_input":"2025-09-16T06:39:42.838419Z","iopub.status.idle":"2025-09-16T06:39:42.857973Z","shell.execute_reply.started":"2025-09-16T06:39:42.838385Z","shell.execute_reply":"2025-09-16T06:39:42.857036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_dicom_series(directory):\n    \"\"\"\n    指定したディレクトリ内のDICOMファイルを読み込み、\n    ImagePositionPatientタグを基にスライス順に並べ替える。\n\n    Args:\n        directory (str): DICOMファイルが含まれるディレクトリへのパス。\n\n    Returns:\n        tuple:\n            - images (np.ndarray): スライス順に並んだ画像の3D配列 (Z, H, W)。\n            - filenames (List[str]): 各スライスに対応するファイルパスのリスト。\n    \"\"\"\n    dicom_files = []\n    for fname in os.listdir(directory):\n        path = os.path.join(directory, fname)\n        if os.path.isfile(path):\n            try:\n                ds = pydicom.dcmread(path, stop_before_pixels=True)\n                if \"ImagePositionPatient\" in ds:\n                    position = ds.ImagePositionPatient[2]  # Z座標\n                    dicom_files.append((position, path))\n            except Exception:\n                continue  # 非DICOMファイルなどは無視\n\n    # Z方向（スライス方向）でソート\n    dicom_files.sort(key=lambda x: x[0])\n\n    # ソートされた順で画像読み込み\n    images = []\n    filenames = []\n    for _, path in dicom_files:\n        ds = pydicom.dcmread(path)\n        image = ds.pixel_array\n        images.append(image)\n        filenames.append(path)\n\n    images = np.stack(images, axis=0)  # (Z, H, W)\n\n    return images, filenames","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:42.858837Z","iopub.execute_input":"2025-09-16T06:39:42.859361Z","iopub.status.idle":"2025-09-16T06:39:42.87366Z","shell.execute_reply.started":"2025-09-16T06:39:42.859335Z","shell.execute_reply":"2025-09-16T06:39:42.872451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pydicom\nimport pandas as pd\n\ndef classify_orientation_and_view(iop, coordinate_system: str = \"LPS\"):\n    \"\"\"\n    iop: [r1,r2,r3,c1,c2,c3]\n    coordinate_system: \"LPS\" (DICOM既定) か \"RAS\"\n    戻り値: (orientation_label in {\"AXIAL\",\"CORONAL\",\"SAGITTAL\"}, anterior_up: bool|None)\n    \"\"\"\n    row = np.array(iop[:3], dtype=float)\n    col = np.array(iop[3:], dtype=float)\n    # 正規化 & 法線\n    def _nz(v):\n        n = np.linalg.norm(v)\n        return v / n if n > 0 else v\n    row = _nz(row); col = _nz(col)\n    normal = _nz(np.cross(row, col))\n\n    axial   = np.array([0, 0, 1], dtype=float)  # Superior 方向\n    coronal = np.array([0, 1, 0], dtype=float)  # Posterior 方向（LPS）\n    sagittal= np.array([1, 0, 0], dtype=float)  # Left 方向（LPS）\n\n    def similarity(a, b):\n        na = np.linalg.norm(a); nb = np.linalg.norm(b)\n        if na == 0 or nb == 0:\n            return -np.inf\n        return abs(float(np.dot(a, b)) / (na * nb))\n\n    scores = {\n        \"AXIAL\":   similarity(normal, axial),\n        \"CORONAL\": similarity(normal, coronal),\n        \"SAGITTAL\":similarity(normal, sagittal),\n    }\n    orientation = max(scores, key=scores.get)\n\n    # 画像の「上」は列方向ベクトル（col）。anteriorベクトルは座標系で変わる\n    up_vector = col\n    if coordinate_system.upper() == \"RAS\":\n        anterior_vec = np.array([0, 1, 0], dtype=float)   # RASのAは +Y\n    else:  # LPS（既定）\n        anterior_vec = np.array([0, -1, 0], dtype=float)  # LPSのAは -Y\n\n    dot = float(np.dot(_nz(up_vector), _nz(anterior_vec)))\n    anterior_up = bool(dot > 0.8)  # しきい値はお好みで\n    return orientation, anterior_up\n\ndef _get_iop_any(ds: pydicom.Dataset):\n    \"\"\"\n    IOP をどこからでも拾う: Shared → PerFrame（先頭で見つかったもの）→ Top-level\n    戻り値: list[6] or None\n    \"\"\"\n    # Shared\n    if 'SharedFunctionalGroupsSequence' in ds:\n        sfg = ds.SharedFunctionalGroupsSequence[0]\n        if 'PlaneOrientationSequence' in sfg:\n            return [float(v) for v in sfg.PlaneOrientationSequence[0].ImageOrientationPatient]\n    # PerFrame（最初に見つかったもの）\n    if 'PerFrameFunctionalGroupsSequence' in ds:\n        for fgs in ds.PerFrameFunctionalGroupsSequence:\n            if 'PlaneOrientationSequence' in fgs:\n                return [float(v) for v in fgs.PlaneOrientationSequence[0].ImageOrientationPatient]\n    # Top-level\n    if 'ImageOrientationPatient' in ds:\n        return [float(v) for v in ds.ImageOrientationPatient]\n    return None\n\ndef _estimate_normal_from_pffgs_positions(ds: pydicom.Dataset):\n    \"\"\"\n    IOPが無いときのフォールバック：PerFrameのIPPから法線（撮影方向）を推定。\n    戻り値: normal(np.ndarray) or None\n    \"\"\"\n    if 'PerFrameFunctionalGroupsSequence' not in ds:\n        return None\n    ipps = []\n    for fgs in ds.PerFrameFunctionalGroupsSequence:\n        if 'PlanePositionSequence' in fgs:\n            ipp = np.array([float(v) for v in fgs.PlanePositionSequence[0].ImagePositionPatient], dtype=float)\n            ipps.append(ipp)\n    if len(ipps) < 2:\n        return None\n    # 単純差分 or PCA の第一主成分でもOK。ここでは端点差分で十分。\n    v = ipps[-1] - ipps[0]\n    nrm = np.linalg.norm(v)\n    if nrm == 0:\n        return None\n    return v / nrm\n\ndef extract_orientation_info(row, coordinate_system: str = \"LPS\"):\n    \"\"\"\n    各行の dicom_dir に対して、orientation と anterior_up を返す。\n    3D（マルチフレーム）にも対応（PFFGS/SharedからIOP、なければIPP推定）。\n    \"\"\"\n    dicom_dir = row[\"SeriesPath\"]\n\n    # まずディレクトリ内のファイルを列挙\n    for fname in sorted(os.listdir(dicom_dir)):\n        fpath = os.path.join(dicom_dir, fname)\n        if not os.path.isfile(fpath):\n            continue\n        try:\n            ds = pydicom.dcmread(fpath, stop_before_pixels=True)\n\n            # 3D（マルチフレーム）でも2Dでも IOP を取得\n            iop = _get_iop_any(ds)\n            if iop is not None:\n                orientation, anterior_up = classify_orientation_and_view(iop, coordinate_system=coordinate_system)\n                return pd.Series({\"orientation_label\": orientation, \"anterior_up\": anterior_up})\n\n            # フォールバック：IOPがどこにも無い → 3Dなら IPP から法線推定して向きだけ判定\n            n = _estimate_normal_from_pffgs_positions(ds)\n            if n is not None:\n                # normal と各基本軸の類似度で AXIAL/CORONAL/SAGITTAL を決定\n                axial   = np.array([0, 0, 1], dtype=float)\n                coronal = np.array([0, 1, 0], dtype=float)\n                sagittal= np.array([1, 0, 0], dtype=float)\n                def sim(a, b):\n                    na = np.linalg.norm(a); nb = np.linalg.norm(b)\n                    return abs(float(np.dot(a, b)) / (na * nb))\n                scores = {\n                    \"AXIAL\": sim(n, axial),\n                    \"CORONAL\": sim(n, coronal),\n                    \"SAGITTAL\": sim(n, sagittal),\n                }\n                orientation = max(scores, key=scores.get)\n                # colベクトルが無いので anterior_up は判定不能\n                return pd.Series({\"orientation_label\": orientation, \"anterior_up\": None})\n\n        except Exception as e:\n            print(f\"スキップ: {fname} - {str(e)}\")\n            continue\n\n    # 何も拾えなかった場合\n    return pd.Series({\"orientation_label\": None, \"anterior_up\": None})\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:42.874482Z","iopub.execute_input":"2025-09-16T06:39:42.874763Z","iopub.status.idle":"2025-09-16T06:39:42.89715Z","shell.execute_reply.started":"2025-09-16T06:39:42.874742Z","shell.execute_reply":"2025-09-16T06:39:42.895881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#df = df[df['SOPInstanceUID'].notna()].reset_index(drop=True)\nseries_root = '/kaggle/input/rsna-intracranial-aneurysm-detection/series'\ndf['SeriesPath'] = df['SeriesInstanceUID'].apply(lambda x: os.path.join(series_root, x))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:42.89811Z","iopub.execute_input":"2025-09-16T06:39:42.898367Z","iopub.status.idle":"2025-09-16T06:39:42.923288Z","shell.execute_reply.started":"2025-09-16T06:39:42.898348Z","shell.execute_reply":"2025-09-16T06:39:42.921818Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[[\"OrientationLabel\", \"AnteriorUp\"]] = df.apply(extract_orientation_info, axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:42.924657Z","iopub.execute_input":"2025-09-16T06:39:42.925231Z","iopub.status.idle":"2025-09-16T06:39:43.293187Z","shell.execute_reply.started":"2025-09-16T06:39:42.925075Z","shell.execute_reply":"2025-09-16T06:39:43.292093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"OrientationLabel\"].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:43.294309Z","iopub.execute_input":"2025-09-16T06:39:43.294674Z","iopub.status.idle":"2025-09-16T06:39:43.30175Z","shell.execute_reply.started":"2025-09-16T06:39:43.294623Z","shell.execute_reply":"2025-09-16T06:39:43.300876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"orientation2idx = {'AXIAL': 0,\n                   'CORONAL': 1,\n                   'SAGITTAL': 2,\n                   None : 3\n                  }\nmodality2idx = {'MRA': 0,\n                'CTA': 1,\n                'MRI T2': 2,\n                'MRI T1post': 3\n               }\ndf['OrientationLabelEncoded'] = df['OrientationLabel'].map(orientation2idx)\ndf['ModalityEncoded'] = df['Modality'].map(modality2idx)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:43.302774Z","iopub.execute_input":"2025-09-16T06:39:43.303076Z","iopub.status.idle":"2025-09-16T06:39:43.322773Z","shell.execute_reply.started":"2025-09-16T06:39:43.303036Z","shell.execute_reply":"2025-09-16T06:39:43.3217Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dcm_paths = list(Path('/kaggle/input/rsna-intracranial-aneurysm-detection/series/1.2.826.0.1.3680043.8.498.10012790035410518400400834395242853657').glob('*'))\ndcm = pydicom.dcmread(dcm_paths[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:43.323815Z","iopub.execute_input":"2025-09-16T06:39:43.324111Z","iopub.status.idle":"2025-09-16T06:39:43.364813Z","shell.execute_reply.started":"2025-09-16T06:39:43.324079Z","shell.execute_reply":"2025-09-16T06:39:43.363879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(dcm.pixel_array[70])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:43.365759Z","iopub.execute_input":"2025-09-16T06:39:43.36613Z","iopub.status.idle":"2025-09-16T06:39:45.815161Z","shell.execute_reply.started":"2025-09-16T06:39:43.366102Z","shell.execute_reply":"2025-09-16T06:39:45.813806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_2d = df[df['ndim']==3].reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:45.816426Z","iopub.execute_input":"2025-09-16T06:39:45.816838Z","iopub.status.idle":"2025-09-16T06:39:45.827145Z","shell.execute_reply.started":"2025-09-16T06:39:45.8168Z","shell.execute_reply":"2025-09-16T06:39:45.826062Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.loc[df['ndim']==2, 'PixelSpacing']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:43:17.693332Z","iopub.execute_input":"2025-09-16T06:43:17.693701Z","iopub.status.idle":"2025-09-16T06:43:17.703225Z","shell.execute_reply.started":"2025-09-16T06:43:17.693675Z","shell.execute_reply":"2025-09-16T06:43:17.702102Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_2d['OrientationLabel'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:45.854194Z","iopub.execute_input":"2025-09-16T06:39:45.854601Z","iopub.status.idle":"2025-09-16T06:39:45.876173Z","shell.execute_reply.started":"2025-09-16T06:39:45.854541Z","shell.execute_reply":"2025-09-16T06:39:45.875145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from iterstrat.ml_stratifiers import MultilabelStratifiedKFold","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:45.87705Z","iopub.execute_input":"2025-09-16T06:39:45.877351Z","iopub.status.idle":"2025-09-16T06:39:45.896806Z","shell.execute_reply.started":"2025-09-16T06:39:45.87733Z","shell.execute_reply":"2025-09-16T06:39:45.895828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = df[['Left Infraclinoid Internal Carotid Artery',\n        'Right Infraclinoid Internal Carotid Artery',\n        'Left Supraclinoid Internal Carotid Artery',\n        'Right Supraclinoid Internal Carotid Artery',\n        'Left Middle Cerebral Artery',\n        'Right Middle Cerebral Artery',\n        'Anterior Communicating Artery',\n        'Left Anterior Cerebral Artery',\n        'Right Anterior Cerebral Artery',\n        'Left Posterior Communicating Artery',\n        'Right Posterior Communicating Artery',\n        'Basilar Tip',\n        'Other Posterior Circulation',\n        'Aneurysm Present',\n        'OrientationLabelEncoded', # axial, sagittal, coronal\n        'ModalityEncoded'\n       ]].to_numpy(dtype=int)\n\nmskf = MultilabelStratifiedKFold(\n    n_splits=5,\n    shuffle=True,\n    random_state=42,\n)\n\n# ── 3) fold 割り当て ──────────────────────────────\ndf[\"fold\"] = -1\nfor fold, (_, val_idx) in enumerate(mskf.split(np.zeros(len(df)), y)):\n    df.loc[val_idx, \"fold\"] = fold","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:45.897804Z","iopub.execute_input":"2025-09-16T06:39:45.898217Z","iopub.status.idle":"2025-09-16T06:39:45.927783Z","shell.execute_reply.started":"2025-09-16T06:39:45.898179Z","shell.execute_reply":"2025-09-16T06:39:45.926449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['PixelSpacing']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:45.92869Z","iopub.execute_input":"2025-09-16T06:39:45.929097Z","iopub.status.idle":"2025-09-16T06:39:45.936661Z","shell.execute_reply.started":"2025-09-16T06:39:45.929065Z","shell.execute_reply":"2025-09-16T06:39:45.935477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.to_csv('train_add_metadata_v5.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:45.937542Z","iopub.execute_input":"2025-09-16T06:39:45.937834Z","iopub.status.idle":"2025-09-16T06:39:46.110024Z","shell.execute_reply.started":"2025-09-16T06:39:45.937812Z","shell.execute_reply":"2025-09-16T06:39:46.108937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Modality'].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T07:48:21.989734Z","iopub.execute_input":"2025-09-16T07:48:21.990031Z","iopub.status.idle":"2025-09-16T07:48:21.997213Z","shell.execute_reply.started":"2025-09-16T07:48:21.990011Z","shell.execute_reply":"2025-09-16T07:48:21.996491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_2d.loc[df_2d['Aneurysm Present']==1, 'OrientationLabel'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:46.111275Z","iopub.execute_input":"2025-09-16T06:39:46.111622Z","iopub.status.idle":"2025-09-16T06:39:46.12017Z","shell.execute_reply.started":"2025-09-16T06:39:46.111567Z","shell.execute_reply":"2025-09-16T06:39:46.119198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_2d.loc[df_2d['Aneurysm Present']==0, 'OrientationLabel'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:46.121327Z","iopub.execute_input":"2025-09-16T06:39:46.122162Z","iopub.status.idle":"2025-09-16T06:39:46.14695Z","shell.execute_reply.started":"2025-09-16T06:39:46.122113Z","shell.execute_reply":"2025-09-16T06:39:46.145952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_2d","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:46.147897Z","iopub.execute_input":"2025-09-16T06:39:46.148181Z","iopub.status.idle":"2025-09-16T06:39:46.178695Z","shell.execute_reply.started":"2025-09-16T06:39:46.14816Z","shell.execute_reply":"2025-09-16T06:39:46.177467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_2d.to_csv('train_2d_extra.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:46.179993Z","iopub.execute_input":"2025-09-16T06:39:46.180386Z","iopub.status.idle":"2025-09-16T06:39:46.198344Z","shell.execute_reply.started":"2025-09-16T06:39:46.180349Z","shell.execute_reply":"2025-09-16T06:39:46.197398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_2d.loc[df_2d['Aneurysm Present'], \"AnteriorUp\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:46.199333Z","iopub.execute_input":"2025-09-16T06:39:46.200081Z","iopub.status.idle":"2025-09-16T06:39:46.222791Z","shell.execute_reply.started":"2025-09-16T06:39:46.200049Z","shell.execute_reply":"2025-09-16T06:39:46.221981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_2d[\"AnteriorUp\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:46.223674Z","iopub.execute_input":"2025-09-16T06:39:46.224043Z","iopub.status.idle":"2025-09-16T06:39:46.239396Z","shell.execute_reply.started":"2025-09-16T06:39:46.224011Z","shell.execute_reply":"2025-09-16T06:39:46.238721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_2d.groupby(['OrientationLabel','Modality'])['Aneurysm Present'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:46.240395Z","iopub.execute_input":"2025-09-16T06:39:46.240786Z","iopub.status.idle":"2025-09-16T06:39:46.263089Z","shell.execute_reply.started":"2025-09-16T06:39:46.240763Z","shell.execute_reply":"2025-09-16T06:39:46.262244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for fold in [0, 1, 2, 3, 4]:\n    print(f'======fold{fold}=======')\n    df_tmp = df[df['fold']==fold]\n    print(df_tmp.groupby(['OrientationLabel','Modality'])['Aneurysm Present'].value_counts())\n    print(df_tmp[['Left Infraclinoid Internal Carotid Artery',\n        'Right Infraclinoid Internal Carotid Artery',\n        'Left Supraclinoid Internal Carotid Artery',\n        'Right Supraclinoid Internal Carotid Artery',\n        'Left Middle Cerebral Artery',\n        'Right Middle Cerebral Artery',\n        'Anterior Communicating Artery',\n        'Left Anterior Cerebral Artery',\n        'Right Anterior Cerebral Artery',\n        'Left Posterior Communicating Artery',\n        'Right Posterior Communicating Artery',\n        'Basilar Tip',\n        'Other Posterior Circulation']].sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:46.264185Z","iopub.execute_input":"2025-09-16T06:39:46.264485Z","iopub.status.idle":"2025-09-16T06:39:46.296846Z","shell.execute_reply.started":"2025-09-16T06:39:46.264459Z","shell.execute_reply":"2025-09-16T06:39:46.295948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.OrientationLabel.unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T06:39:46.297719Z","iopub.execute_input":"2025-09-16T06:39:46.298012Z","iopub.status.idle":"2025-09-16T06:39:46.304368Z","shell.execute_reply.started":"2025-09-16T06:39:46.297993Z","shell.execute_reply":"2025-09-16T06:39:46.303566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}