{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":18647,"databundleVersionId":1126921}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\n\n# 读取 train.csv\ndf = pd.read_csv('/kaggle/input/competitions/prostate-cancer-grade-assessment/train.csv')\n\n# 筛选条件：isup_grade > 0 且 data_provider == 'radboud'\ndf_filtered = df[(df['isup_grade'] > 0) & (df['data_provider'] == 'radboud')]\n\nprint(f\"原始数据行数: {len(df)}\")\nprint(f\"筛选后行数: {len(df_filtered)}\")\nprint(df_filtered.head())","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-12T13:37:37.258502Z","iopub.execute_input":"2026-04-12T13:37:37.259531Z","iopub.status.idle":"2026-04-12T13:37:37.288057Z","shell.execute_reply.started":"2026-04-12T13:37:37.259493Z","shell.execute_reply":"2026-04-12T13:37:37.287023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install imagecodecs -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T13:37:40.009215Z","iopub.execute_input":"2026-04-12T13:37:40.010358Z","iopub.status.idle":"2026-04-12T13:37:46.523055Z","shell.execute_reply.started":"2026-04-12T13:37:40.010309Z","shell.execute_reply":"2026-04-12T13:37:46.521615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tifffile\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# 提取第一行数据\nfirst_row = df_filtered.iloc[0]\nimage_id = first_row['image_id']\nprint(f\"image_id: {image_id}\")\nprint(f\"isup_grade: {first_row['isup_grade']}, data_provider: {first_row['data_provider']}\")\n\n# 拼接 mask 路径\nmask_path = f'/kaggle/input/competitions/prostate-cancer-grade-assessment/train_label_masks/{image_id}_mask.tiff'\n\n# 读取并可视化\nwith tifffile.TiffFile(mask_path) as tif:\n    image = tif.series[0].levels[-1].asarray()\n    print(f\"图像尺寸: {image.shape}\")\n    print(f\"唯一值 (Gleason Pattern): {np.unique(image)}\")\n\nplt.figure(figsize=(10, 10))\nplt.imshow(image * 50, cmap='jet')\nplt.colorbar(label='Gleason Pattern ID')\nplt.title(f\"Mask: {image_id}\\nisup_grade={first_row['isup_grade']}, provider={first_row['data_provider']}\")\nplt.axis('off')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T13:37:50.239104Z","iopub.execute_input":"2026-04-12T13:37:50.240023Z","iopub.status.idle":"2026-04-12T13:37:51.002836Z","shell.execute_reply.started":"2026-04-12T13:37:50.239981Z","shell.execute_reply":"2026-04-12T13:37:51.001741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tifffile\nimport numpy as np\nimport matplotlib.pyplot as plt\n\npatch_size_l0 = 1536\npatch_size_l1 = 384\nstep = 64\n\n# mask 路径和对应的原图路径\nimage_path = mask_path.replace('train_label_masks', 'train_images').replace('_mask.tiff', '.tiff')\nprint(f\"mask 路径:  {mask_path}\")\nprint(f\"原图路径:  {image_path}\")\n\nwith tifffile.TiffFile(mask_path) as tif:\n    levels = tif.series[0].levels\n    small = levels[1].asarray()\n    H_s, W_s = small.shape[:2]\n    print(f\"\\nlevel1 尺寸: {H_s} x {W_s}\")\n\n    best_score = -np.inf\n    best_pos = None\n    best_stats = None\n\n    for y in range(0, H_s - patch_size_l1, step):\n        for x in range(0, W_s - patch_size_l1, step):\n            patch = small[y:y+patch_size_l1, x:x+patch_size_l1]\n            if patch.ndim == 3:\n                patch = patch[..., 0]\n\n            total = patch.size\n            unique_vals = set(np.unique(patch))\n            counts = {v: np.sum(patch == v) for v in range(6)}\n            ratios = {v: counts[v] / total for v in range(6)}\n            cancer_ratio = ratios[3] + ratios[4] + ratios[5]\n            normal_ratio = ratios[1] + ratios[2]\n\n            if ratios[0] > 0.20:            continue\n            if not unique_vals & {3, 4, 5}: continue\n            if not unique_vals & {1, 2}:    continue\n            if cancer_ratio < 0.15:         continue\n            if normal_ratio < 0.20:         continue\n\n            score = cancer_ratio * 2.0 + normal_ratio * 1.5 - ratios[0] * 4.0\n            if score > best_score:\n                best_score = score\n                best_pos = (y, x)\n                best_stats = {\n                    \"unique\":       sorted(unique_vals),\n                    \"cancer_ratio\": round(cancer_ratio, 3),\n                    \"normal_ratio\": round(normal_ratio, 3),\n                    \"bg_ratio\":     round(ratios[0], 3),\n                }\n\n    print(f\"最优位置 (level1): {best_pos}，得分: {best_score:.4f}\")\n    print(f\"统计: {best_stats}\")\n\n    if best_pos is None:\n        print(\"⚠️ 没有找到满足条件的 patch\")\n    else:\n        y0 = best_pos[0] * 4\n        x0 = best_pos[1] * 4\n        print(f\"level0 对应位置: ({y0}, {x0})\")\n\n        # 裁 mask\n        mask_full = levels[0].asarray()\n        if mask_full.ndim == 3:\n            mask_full = mask_full[..., 0]\n        best_mask = mask_full[y0:y0+patch_size_l0, x0:x0+patch_size_l0]\n\n# 裁原图（单独开文件句柄）\nwith tifffile.TiffFile(image_path) as tif_img:\n    img_full = tif_img.series[0].levels[0].asarray()\n    best_img = img_full[y0:y0+patch_size_l0, x0:x0+patch_size_l0]\n    print(f\"原图 patch: {best_img.shape}, dtype: {best_img.dtype}\")\n\n# 可视化：三图并排\nfig, axes = plt.subplots(1, 3, figsize=(20, 7))\n\n# 左：原始 H&E 图像\naxes[0].imshow(best_img)\naxes[0].set_title(f\"H&E 原图\\n@ level0 ({y0}, {x0})\")\naxes[0].axis('off')\n\n# 中：mask 标注\ncmap = plt.cm.get_cmap('jet', 6)\nim = axes[1].imshow(best_mask, cmap=cmap, vmin=0, vmax=5)\nplt.colorbar(im, ax=axes[1], ticks=range(6),\n             label='0=BG 1=Stroma 2=Benign 3=G3 4=G4 5=G5')\naxes[1].set_title(f\"Mask 标注\\ncancer={best_stats['cancer_ratio']:.1%}  normal={best_stats['normal_ratio']:.1%}\")\naxes[1].axis('off')\n\n# 右：原图 + mask 叠加\noverlay = best_img.copy().astype(float) / 255.0\ncolor_map = {\n    0: [0,   0,   0  ],   # 背景：不叠加\n    1: [0.0, 0.8, 0.0],   # stroma：绿\n    2: [0.0, 0.5, 1.0],   # 良性：蓝\n    3: [1.0, 1.0, 0.0],   # G3：黄\n    4: [1.0, 0.4, 0.0],   # G4：橙\n    5: [1.0, 0.0, 0.0],   # G5：红\n}\nblended = overlay.copy()\nalpha = 0.45\nfor label, color in color_map.items():\n    if label == 0:\n        continue\n    mask_region = best_mask == label\n    for c in range(3):\n        blended[mask_region, c] = (\n            overlay[mask_region, c] * (1 - alpha) + color[c] * alpha\n        )\naxes[2].imshow(blended)\naxes[2].set_title(\"H&E + Mask 叠加\")\naxes[2].axis('off')\n\nplt.suptitle(f\"{image_id}  |  unique labels: {best_stats['unique']}\", fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T13:37:54.73613Z","iopub.execute_input":"2026-04-12T13:37:54.737283Z","iopub.status.idle":"2026-04-12T13:38:04.39066Z","shell.execute_reply.started":"2026-04-12T13:37:54.737242Z","shell.execute_reply":"2026-04-12T13:38:04.38922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tifffile\nimport numpy as np\nimport os\nfrom PIL import Image\nfrom tqdm import tqdm\n\n# 保存目录\nsave_dir = '/kaggle/working/patches'\nos.makedirs(f'{save_dir}/images', exist_ok=True)\nos.makedirs(f'{save_dir}/masks',  exist_ok=True)\n\npatch_size_l0 = 1536\npatch_size_l1 = 384\nstep = 64\nmax_save = 1000\n\nsaved = 0\nskipped = 0\n\nfor _, row in tqdm(df_filtered.iterrows(), total=min(len(df_filtered), max_save * 3)):\n    if saved >= max_save:\n        break\n\n    image_id  = row['image_id']\n    mask_path = f'/kaggle/input/competitions/prostate-cancer-grade-assessment/train_label_masks/{image_id}_mask.tiff'\n    image_path = f'/kaggle/input/competitions/prostate-cancer-grade-assessment/train_images/{image_id}.tiff'\n\n    # 跳过不存在的文件\n    if not os.path.exists(mask_path) or not os.path.exists(image_path):\n        skipped += 1\n        continue\n\n    try:\n        # ── 在 level1 中找最优 patch ──────────────────────\n        best_pos = None\n        best_score = -np.inf\n        best_stats = None\n\n        with tifffile.TiffFile(mask_path) as tif:\n            levels = tif.series[0].levels\n            small = levels[1].asarray()\n            H_s, W_s = small.shape[:2]\n\n            # level1 可能比 patch 还小，直接跳过\n            if H_s < patch_size_l1 or W_s < patch_size_l1:\n                skipped += 1\n                continue\n\n            for y in range(0, H_s - patch_size_l1, step):\n                for x in range(0, W_s - patch_size_l1, step):\n                    patch = small[y:y+patch_size_l1, x:x+patch_size_l1]\n                    if patch.ndim == 3:\n                        patch = patch[..., 0]\n\n                    total = patch.size\n                    unique_vals = set(np.unique(patch))\n                    counts = {v: np.sum(patch == v) for v in range(6)}\n                    ratios = {v: counts[v] / total for v in range(6)}\n                    cancer_ratio = ratios[3] + ratios[4] + ratios[5]\n                    normal_ratio = ratios[1] + ratios[2]\n\n                    if ratios[0] > 0.20:            continue\n                    if not unique_vals & {3, 4, 5}: continue\n                    if not unique_vals & {1, 2}:    continue\n                    if cancer_ratio < 0.15:         continue\n                    if normal_ratio < 0.20:         continue\n\n                    score = cancer_ratio * 2.0 + normal_ratio * 1.5 - ratios[0] * 4.0\n                    if score > best_score:\n                        best_score = score\n                        best_pos = (y, x)\n                        best_stats = {\n                            \"cancer_ratio\": round(cancer_ratio, 3),\n                            \"normal_ratio\": round(normal_ratio, 3),\n                            \"bg_ratio\":     round(ratios[0], 3),\n                        }\n\n            # 没找到合适 patch，跳过\n            if best_pos is None:\n                skipped += 1\n                continue\n\n            y0 = best_pos[0] * 4\n            x0 = best_pos[1] * 4\n\n            # 裁 mask\n            mask_full = levels[0].asarray()\n            if mask_full.ndim == 3:\n                mask_full = mask_full[..., 0]\n            best_mask = mask_full[y0:y0+patch_size_l0, x0:x0+patch_size_l0]\n\n        # ── 裁原图 ────────────────────────────────────────\n        with tifffile.TiffFile(image_path) as tif_img:\n            img_full = tif_img.series[0].levels[0].asarray()\n            best_img = img_full[y0:y0+patch_size_l0, x0:x0+patch_size_l0]\n\n        # ── 保存 ──────────────────────────────────────────\n        fname = f'{image_id}_y{y0}_x{x0}'\n        Image.fromarray(best_img).save(f'{save_dir}/images/{fname}.png')\n        Image.fromarray(best_mask.astype(np.uint8)).save(f'{save_dir}/masks/{fname}.png')\n\n        saved += 1\n\n    except Exception as e:\n        print(f\"⚠️ {image_id} 处理失败: {e}\")\n        skipped += 1\n        continue\n\nprint(f\"\\n完成：保存 {saved} 张，跳过 {skipped} 张\")\nprint(f\"保存路径: {save_dir}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T13:38:37.177068Z","iopub.execute_input":"2026-04-12T13:38:37.177534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\nimage_dir = '/kaggle/working/patches/images'\nmask_dir  = '/kaggle/working/patches/masks'\n\n# 取前6张展示\nfnames = sorted(os.listdir(image_dir))[:6]\n\nfig, axes = plt.subplots(len(fnames), 3, figsize=(18, 6 * len(fnames)))\n\ncolor_map = {\n    1: [0.0, 0.8, 0.0],   # stroma：绿\n    2: [0.0, 0.5, 1.0],   # 良性：蓝\n    3: [1.0, 1.0, 0.0],   # G3：黄\n    4: [1.0, 0.4, 0.0],   # G4：橙\n    5: [1.0, 0.0, 0.0],   # G5：红\n}\n\nfor i, fname in enumerate(fnames):\n    img  = np.array(Image.open(f'{image_dir}/{fname}'))\n    mask = np.array(Image.open(f'{mask_dir}/{fname}'))\n\n    # 叠加\n    overlay = img.astype(float) / 255.0\n    blended = overlay.copy()\n    alpha = 0.45\n    for label, color in color_map.items():\n        region = mask == label\n        for c in range(3):\n            blended[region, c] = overlay[region, c] * (1 - alpha) + color[c] * alpha\n\n    # 统计\n    total = mask.size\n    cancer_ratio = np.sum(np.isin(mask, [3,4,5])) / total\n    normal_ratio = np.sum(np.isin(mask, [1,2]))   / total\n    bg_ratio     = np.sum(mask == 0)               / total\n\n    axes[i][0].imshow(img)\n    axes[i][0].set_title(f\"H&E\\n{fname[:30]}\", fontsize=9)\n    axes[i][0].axis('off')\n\n    cmap = plt.cm.get_cmap('jet', 6)\n    im = axes[i][1].imshow(mask, cmap=cmap, vmin=0, vmax=5)\n    plt.colorbar(im, ax=axes[i][1], ticks=range(6),\n                 label='0=BG 1=Stroma 2=Benign 3=G3 4=G4 5=G5')\n    axes[i][1].set_title(f\"Mask\\ncancer={cancer_ratio:.1%} normal={normal_ratio:.1%} bg={bg_ratio:.1%}\", fontsize=9)\n    axes[i][1].axis('off')\n\n    axes[i][2].imshow(blended)\n    axes[i][2].set_title(\"叠加\", fontsize=9)\n    axes[i][2].axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T02:28:39.113972Z","iopub.execute_input":"2026-04-02T02:28:39.114294Z","iopub.status.idle":"2026-04-02T02:28:47.818748Z","shell.execute_reply.started":"2026-04-02T02:28:39.114266Z","shell.execute_reply":"2026-04-02T02:28:47.817778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nshutil.make_archive('/kaggle/working/patches_1000', 'zip', '/kaggle/working/patches')\nprint(\"打包完成：/kaggle/working/patches_1000.zip\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T02:25:24.967637Z","iopub.status.idle":"2026-04-02T02:25:24.967987Z","shell.execute_reply.started":"2026-04-02T02:25:24.967845Z","shell.execute_reply":"2026-04-02T02:25:24.967865Z"}},"outputs":[],"execution_count":null}]}